{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_ff5ef05ac79acfe1caaf55389f73b201f237de6247aac07ba2926b023ab85045","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_ff5ef05ac79acfe1caaf55389f73b201f237de6247aac07ba2926b023ab85045","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"78f3086fcdbb4e02f095366d748280dff3d397ab55b7e34ddb887655d573ce72","published":"Mon, 20 Jul 2026 00:00:00 -0400","receipt_hash":"78f3086fcdbb4e02f095366d748280dff3d397ab55b7e34ddb887655d573ce72","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"78f3086fcdbb4e02f095366d748280dff3d397ab55b7e34ddb887655d573ce72","observed_at":"2026-07-20T04:43:09.641409Z","parent_run_hash":"0fd83663f0f57da59b26313ca1a35283a3e9f06e3165d3e143d26f7174743aca","published":"Mon, 20 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.15778v1 Announce Type: cross \nAbstract: Video Large Language Models (Video LLMs) have made significant advancements in various video understanding tasks. However, long-video scenarios remain challenging due to the tension between limited visual token budgets and the need to capture multiple key events. Existing approaches typically process long videos in two stages, i.e., i) select keyframes and ii) perform detailed perception, which exhibit limitations: they lack a modular mechanism for adaptive capacity allocation and self-correction, resulting in unreliable modeling. To tackle these challenges, we propose MoD-VLLM, a novel Modularized Dynamic-Granularity Video LLM framework for multi-event long video understanding, which unifies temporal grounding and semantic understanding iteratively and self-reflectively. Specifically, we propose a Positive-Negative Video Segments Grounding module and a Modularized Dynamic-Granularity Reflection module, which form a closed loop to prog","title":"Modularized Dynamic-Granularity Video LLM for Multi-Event Long Video Understanding","url":"https://arxiv.org/abs/2607.15778","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.15778v1 Announce Type: cross \nAbstract: Video Large Language Models (Video LLMs) have made significant advancements in various video understanding tasks. However, long-video scenarios remain challenging due to the tension between limited visual token budgets and the need to capture multiple key events. Existing approaches typically process long videos in two stages, i.e., i) select keyframes and ii) perform detailed perception, which exhibit limitations: they lack a modular mechanism for adaptive capacity allocation and self-correction, resulting in unreliable modeling. To tackle these challenges, we propose MoD-VLLM, a novel Modularized Dynamic-Granularity Video LLM framework for multi-event long video understanding, which unifies temporal grounding and semantic understanding iteratively and self-reflectively. Specifically, we propose a Positive-Negative Video Segments Grounding module and a Modularized Dynamic-Granularity Reflection module, which form a closed loop to prog","title":"Modularized Dynamic-Granularity Video LLM for Multi-Event Long Video Understanding","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-20T04:43:09Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.15778"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:2a2471d5de9ba037fd01d548193140f20d8a232668a3ef7ba4d9aa544c30de574fdf9893bbb8ec593d90470f3301c6ddecadccf4f9283145431bd2fc511af30c","signer":"crovia.substrate","subject":{"observed_at":"2026-07-20T04:43:09Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.15778"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"a58d3b3890ff3f957b2bc3bd100262739f62d83d216b0c36426482055c8c269a","leaf_index":333295,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"58fcf440a8bb0371e8102910f7dada1a66af823f2726ea24d330fdb662275841","side":"left"},{"sibling":"58db31d28dec0c9af7669d97c62c17d9f126895b18e4f3325b880e4d2490bbda","side":"left"},{"sibling":"088716cf2099b1bb6121e9e2de37a52edb8879c658c331715ce3a50c04ce2e0e","side":"left"},{"sibling":"41507d4f0627629d80257cc6d445c55a4b2d57d958d00c3bfd525e2e5c31a176","side":"left"},{"sibling":"758c5f596693405fc5fef214df56ab7784d8c080973814df55ef32131cf4c328","side":"right"},{"sibling":"31344f2819390838d1f446c7b73dcdde092475dfd3f66f0caadf2d2e1ba1a927","side":"left"},{"sibling":"5dc3f0df8fd6e4cf655b15e7355878576934a564ddee081917870d5b8bcd2ec0","side":"left"},{"sibling":"5e895feb6aeddb3dac892d493e1de271cf5bb255124626b8ae869ebb57ff0c74","side":"left"},{"sibling":"089fcaa13823a284a9c3064c9de337dded31410d6f89ef3761586a5e77c18ff1","side":"left"},{"sibling":"b0d667e95f8746ac275ad4169634f2e9442cd7e445099c6355bed5e0b9e27cbf","side":"right"},{"sibling":"dedd2da92d9447ddf1b1db68fe20109a426ed18359861ed746907ac820021a8d","side":"left"},{"sibling":"a1c43cc7cd9c775fac33940ee5124aece01596733f003fc53743f43483f9f597","side":"right"},{"sibling":"b5ad3eafd7eeb74c063261356fdd9bf6059ee6d0bf1e3c70e60731b394a5536e","side":"left"},{"sibling":"93e399d152203db688c6a5a58d25131205603504f5b79123a1f2b5a5ed9c1e54","side":"right"},{"sibling":"b6e0cad7f6eb9107f0edd276f1a9942635d8cd6d60d2a97e7daac08b110dc209","side":"right"},{"sibling":"80ec062e7e625dc3f9bb5865cb5198696bbec2608e48abae5670677b90695899","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"0aced6f0c9dec3e6cc9e89b68b70f5f8ce7e1eb13606d92db1917b76e57393c7","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":333540,"merkle_root":"ee60f62b8a724dd9bde638d638caf32cefec4440f832018b457ff47a0ec56a8c","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260720T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-20T05:38:36Z","sig_algorithm":"ed25519","signature":"82a3787e628bfab19c377d875220e1aaedfc498736c545f4708a1b887e8398afdf306995994493a36c864ff7139a2d436b905ce7081aaa540789aa4f707dc800","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_ff5ef05ac79acfe1caaf55389f73b201f237de6247aac07ba2926b023ab85045"}}