{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_9db7b354e6a1a3be1b6de5e79ac7b07991bb4f258372b132f7918a90b6b4033e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_9db7b354e6a1a3be1b6de5e79ac7b07991bb4f258372b132f7918a90b6b4033e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"ac33886c4e8debe766ffaef4882121b4a90284fc47c1556bf358dd860bc97f13","published":"Thu, 21 May 2026 00:00:00 -0400","receipt_hash":"ac33886c4e8debe766ffaef4882121b4a90284fc47c1556bf358dd860bc97f13","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"ac33886c4e8debe766ffaef4882121b4a90284fc47c1556bf358dd860bc97f13","observed_at":"2026-05-21T04:43:31.073835Z","parent_run_hash":"5047c4a56b18dd764734c0da16c60af14d40680f3bac4bcad4e9983899d001f9","published":"Thu, 21 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.19846v2 Announce Type: cross \nAbstract: Vision-Language Models (VLMs) have demonstrated remarkable capabilities in general video understanding, yet they often struggle with the fine-grained comprehension crucial for real-world applications requiring nuanced interpretation of human actions and interactions. While some recent human-centric benchmarks evaluate aspects of model behaviour such as fairness/ethics, emotion perception, and broader human-centric metrics, they do not combine long-form videos, very dense QA coverage, and frame-level spatial/temporal grounding at scale. To bridge this gap, we introduce FineBench, a human-centric video question answering (VQA) benchmark specifically designed to assess fine-grained understanding. FineBench comprises 199,420 multiple-choice QA pairs densely annotated across 64 long-form videos (15 minutes each), focusing on detailed person movement, person interaction, and object manipulation, including compositional actions. Our extensive","title":"FineBench: Benchmarking and Enhancing Vision-Language Models for Fine-grained Human Activity Understanding","url":"https://arxiv.org/abs/2605.19846","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.19846v2 Announce Type: cross \nAbstract: Vision-Language Models (VLMs) have demonstrated remarkable capabilities in general video understanding, yet they often struggle with the fine-grained comprehension crucial for real-world applications requiring nuanced interpretation of human actions and interactions. While some recent human-centric benchmarks evaluate aspects of model behaviour such as fairness/ethics, emotion perception, and broader human-centric metrics, they do not combine long-form videos, very dense QA coverage, and frame-level spatial/temporal grounding at scale. To bridge this gap, we introduce FineBench, a human-centric video question answering (VQA) benchmark specifically designed to assess fine-grained understanding. FineBench comprises 199,420 multiple-choice QA pairs densely annotated across 64 long-form videos (15 minutes each), focusing on detailed person movement, person interaction, and object manipulation, including compositional actions. Our extensive","title":"FineBench: Benchmarking and Enhancing Vision-Language Models for Fine-grained Human Activity Understanding","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-21T04:43:31Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.19846"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:751c78c91b5fe8e52bb7d8741e778bf0fe3b8c32c3b1bf7399b3cc32b7f0e0c9e8b134680a9cae234e03d310107fe82304f176bd6311df6ea1fceba2db96040e","signer":"crovia.substrate","subject":{"observed_at":"2026-05-21T04:43:31Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.19846"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1db2256afebcdfd25cce95a6dd7949065220d88e21ef96acb0ba7a00445ef510","leaf_index":146071,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"b73370b88adeaeb0559e232c095a466945993db30fb4dab4cb00180207516949","side":"left"},{"sibling":"26c24cfacfed58f17b05927a2258447f6afd24009bdf6fc5ba961ec4ece4d1a6","side":"left"},{"sibling":"2972d8e705f9a395a718c526995c23594f54daf600c882f9d25b1616de57c307","side":"left"},{"sibling":"f3e6b67ad8cfb471fab17c8f2fb4c62cef62429fa6ec27cb7377aea125138ec7","side":"right"},{"sibling":"b957aa3392a47d83fbb1a6819563adb3ad6dd0f86a5aafbb38a1d68e952d2645","side":"left"},{"sibling":"b2a2594970499ef0b84e8624114d226ef874a95c1d9b9a6de0646e0d7fca0d4c","side":"right"},{"sibling":"cfdfcdf15c10b7f30f2c03e430270fe58f21c87e1dd8b50b3e107416147771ca","side":"right"},{"sibling":"e2d788db0d8449372377479a631b42a652a64ed491d720f74dda5781d988ae76","side":"left"},{"sibling":"f39a7335aa6b82bde60c7b432781bd9e0a9e8ae7bfc7bd50eeb0fe7acdf83691","side":"right"},{"sibling":"3d048271f4871de655c4d4e2d83c1a269789f9e8b4e12a7f789beea57480c8a4","side":"left"},{"sibling":"14b1a643ffa50eeeb494fd3a15fd505a15e79133c339d915d5cac16c81067c49","side":"right"},{"sibling":"8c417ae33891ef4f8dbdef16d48b6a102716fb1ddbdcae54ff85dcedb4f657d4","side":"left"},{"sibling":"3e4df6e7457cecbf36f350375e72dcab336a3984422e4c406ef809e4e2944e96","side":"left"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"be08fedc4e72a6fb56606f66812fae7317e09690b9acb18385f4ab117a981238","side":"right"},{"sibling":"0534329a7475dc9df51998c83c16892126679dade0fa34182f21e869599386c7","side":"right"},{"sibling":"2d24720928ead0e7670650eb55f558c4f20e4c18df376f47ba72cfa8cf0ed344","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":147301,"merkle_root":"08903d7159c3b38eeeeafc09eab15139ea417f1d94f02f1fbc87296b37db840a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260521T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-21T18:37:33Z","sig_algorithm":"ed25519","signature":"905f2924632dfa2970c8690285f5b5d4a1d891d0e0ef1cbc404ebec2fd937215ac768e16f0a9f28b18977a55ae0bfd226db6833ae7ef588729054117d2da7303","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_9db7b354e6a1a3be1b6de5e79ac7b07991bb4f258372b132f7918a90b6b4033e"}}