{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_704a38267837ad36f96cc7670871442d10f7ceefb656a95083633f2779ba9079","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_704a38267837ad36f96cc7670871442d10f7ceefb656a95083633f2779ba9079","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"207cbb51ee65458c879443abc13e40bbf087c64cff95769222065531a02469fc","published":"Mon, 18 May 2026 00:00:00 -0400","receipt_hash":"207cbb51ee65458c879443abc13e40bbf087c64cff95769222065531a02469fc","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"207cbb51ee65458c879443abc13e40bbf087c64cff95769222065531a02469fc","observed_at":"2026-05-18T04:43:11.219741Z","parent_run_hash":"a8aad7414ebb6b75c726f09cd673410576a7f87e191fbdb9ddac99e9b2b95a05","published":"Mon, 18 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.16079v1 Announce Type: cross \nAbstract: Large Vision-Language Models (LVLMs) have shown significant progress in video understanding, yet they face substantial challenges in tasks requiring precise spatiotemporal localization at the instance level. Existing methods primarily rely on text prompts for human-model interaction, but these prompts struggle to provide precise spatial and temporal references, resulting in poor user experience. Furthermore, current approaches typically decouple visual perception from language reasoning, centering reasoning around language rather than visual content, which limits the model's ability to proactively perceive fine-grained visual evidence. To address these challenges, we propose VideoSeeker, a novel paradigm for instance-level video understanding through visual prompts. VideoSeeker seamlessly integrates agentic reasoning with instance-level video understanding tasks, enabling the model to proactively perceive and retrieve relevant video se","title":"VideoSeeker: Incentivizing Instance-level Video Understanding via Native Agentic Tool Invocation","url":"https://arxiv.org/abs/2605.16079","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.16079v1 Announce Type: cross \nAbstract: Large Vision-Language Models (LVLMs) have shown significant progress in video understanding, yet they face substantial challenges in tasks requiring precise spatiotemporal localization at the instance level. Existing methods primarily rely on text prompts for human-model interaction, but these prompts struggle to provide precise spatial and temporal references, resulting in poor user experience. Furthermore, current approaches typically decouple visual perception from language reasoning, centering reasoning around language rather than visual content, which limits the model's ability to proactively perceive fine-grained visual evidence. To address these challenges, we propose VideoSeeker, a novel paradigm for instance-level video understanding through visual prompts. VideoSeeker seamlessly integrates agentic reasoning with instance-level video understanding tasks, enabling the model to proactively perceive and retrieve relevant video se","title":"VideoSeeker: Incentivizing Instance-level Video Understanding via Native Agentic Tool Invocation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-18T04:43:11Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.16079"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1aadc6e464210c8cdcffd8e2b1efcc96e08fa1ec37726dcc050366a6923b39cc15c1b4e6544752aa3be524da4945cae9c7686774d55a70340f4f4acfdf4a5702","signer":"crovia.substrate","subject":{"observed_at":"2026-05-18T04:43:11Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.16079"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"09f1c949f24d209da856859675271d7b997929fe81a23c1c9c424895770fc3e5","leaf_index":140676,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7c9715ab9f9a552342142a322718fbd12ad4157973eabd1e42de5e30d8816f5d","side":"right"},{"sibling":"ef16f7a4c596139080b358f9f1a29785634b8a05afbf2fb25c0ebbdbc8c347e5","side":"right"},{"sibling":"5cce6779e74890c6be433dc80b994906254fefb26c61138cb272d84e340f5c34","side":"left"},{"sibling":"2aa9e74a06ea77cb5df7d56fe22ded9719f1ca4f3f256c569470b69cf29cf3a4","side":"right"},{"sibling":"58683ddf35d7c3cb9115208c0129b2442e62481772681c4cb2b70868d4f555a7","side":"right"},{"sibling":"ccfc733d99355b280b7859d93aaff9b798cfb41eee6381849bd91a1ab4a17966","side":"right"},{"sibling":"c506700de9fc9cdf32017abe38d10a05938325fe576d9c37ea5fba2a436c2162","side":"right"},{"sibling":"30c60828b6b0ade79197e585b88c06ccf4f348c1e62fbf6710d89ae2c2e9dcfb","side":"left"},{"sibling":"6bedf73520cf3dd8758d8bdedf3be245de9aea97abd42934aae25539176ae1b2","side":"left"},{"sibling":"07abc3bad689e74e6304772503dc9372a118e6f66883b8e88c43414efddac063","side":"right"},{"sibling":"28b78fb112bcf26b6801664db97eb8f52a9bccbf0a7ae6766e11845d443692df","side":"left"},{"sibling":"68d0a4634c1460a19c92edd9480df3aa733b814463e7420d1e14471bf61b2f83","side":"right"},{"sibling":"8af64f275b862349aa3bbb9d5cd7fa9a7fdd5620af3bf1b36b2a4519b0b53bdf","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"b98c2afadb358e5387e88f19588f8343a81b488d9b44a6f7e57a032db3a1b030","side":"right"},{"sibling":"11b0c1591747f09f7c8971a6caa19befcd81317ca9dfd417b143234df4e10c79","side":"right"},{"sibling":"87206f3bcc342797c990d87f7235c01f78d32ca59cfaf8ad18d71afc879ba477","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":140892,"merkle_root":"6cca56ead155990456b8a014cc50bddbe710f409b26e3d1bfa6fb12b0bfcf6bf","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260518T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-18T05:37:30Z","sig_algorithm":"ed25519","signature":"1e1135f7595f79b14fb11f5fa81a2e17ad31b11b44b427a5e40a7d511cd86447daf492babd368ab571cf26404c8c74c450d460130fca4b064eb2760367489a0f","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_704a38267837ad36f96cc7670871442d10f7ceefb656a95083633f2779ba9079"}}