{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b848631fc239498dfb6daf26aa8634d59b9d2331102ce0e6acc35a1348820ad2","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b848631fc239498dfb6daf26aa8634d59b9d2331102ce0e6acc35a1348820ad2","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f2c252c6914c67d8356c1aa75c92c50fc4bce1db8bde9f7f411320c0b0bd6980","published":"Thu, 28 May 2026 00:00:00 -0400","receipt_hash":"f2c252c6914c67d8356c1aa75c92c50fc4bce1db8bde9f7f411320c0b0bd6980","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f2c252c6914c67d8356c1aa75c92c50fc4bce1db8bde9f7f411320c0b0bd6980","observed_at":"2026-05-28T04:43:38.862500Z","parent_run_hash":"58f8b4a134069e0a15ea3949252489597eb86dd27c9ca3fb15c6fb838ce49ef3","published":"Thu, 28 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.28160v1 Announce Type: new \nAbstract: Existing multimodal reasoning approaches predominantly follow two paradigms: converting visual inputs into text prior to reasoning, or performing end-to-end reasoning within a unified vision-language representation space. Despite their empirical progress, both paradigms suffer from fundamental structural limitations. The former relies on static visual-to-text conversion, which tends to compress and lose fine-grained visual details. The latter is prone to linguistic dominance induced by joint optimization and attention mechanisms, leading to systematically weakened faithfulness to visual evidence during reasoning. In this work, we argue that a central challenge is how and when visual evidence is introduced into the reasoning process. Motivated by this insight, we propose CSMR, a multimodal reasoning framework in which a language model controls the reasoning process by deciding when to invoke an independent visual perception module to acqu","title":"Look on Demand: A Cognitive Scheduling Framework for Visual Evidence Acquisition in Multimodal Reasoning","url":"https://arxiv.org/abs/2605.28160","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.28160v1 Announce Type: new \nAbstract: Existing multimodal reasoning approaches predominantly follow two paradigms: converting visual inputs into text prior to reasoning, or performing end-to-end reasoning within a unified vision-language representation space. Despite their empirical progress, both paradigms suffer from fundamental structural limitations. The former relies on static visual-to-text conversion, which tends to compress and lose fine-grained visual details. The latter is prone to linguistic dominance induced by joint optimization and attention mechanisms, leading to systematically weakened faithfulness to visual evidence during reasoning. In this work, we argue that a central challenge is how and when visual evidence is introduced into the reasoning process. Motivated by this insight, we propose CSMR, a multimodal reasoning framework in which a language model controls the reasoning process by deciding when to invoke an independent visual perception module to acqu","title":"Look on Demand: A Cognitive Scheduling Framework for Visual Evidence Acquisition in Multimodal Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-28T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.28160"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:972dba0c3e59b8bae7aa8461efc3f1dc9afe21fd8e7a6189da45ba909ccf2fdd734fbdf3ee81a7287576749c7fc4da46d590c7505251a76296605d596f944308","signer":"crovia.substrate","subject":{"observed_at":"2026-05-28T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.28160"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"acc58757d54c92796e93592d0660dc437a65981bb42cf8ecd76be505a357b511","leaf_index":155652,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6c7a1f422562ff0e701df5d030dbbe927f1954247eaf77072ad320f70a36bcd6","side":"right"},{"sibling":"63da5cb983ca1a27abb8a508e9938473b5bc1b1b436a0704ebb54044582a6bd9","side":"right"},{"sibling":"12f897a876534a9235a54af76b31e7641351c58f65cf6956852115217240fcc4","side":"left"},{"sibling":"aa9ed634698696e2bcbfc81d955532a862dd6e48ecb1c6246592601ef63f7594","side":"right"},{"sibling":"aecee5d7c75c6cf838dea352e16721c8da83b5bd8c49cf4ab9760d6f1beda811","side":"right"},{"sibling":"ed8d8b6288b10a26a2d1fd017d1913ca7ea0840db020fcdf5375721f1856f0f8","side":"right"},{"sibling":"36a1078bdcaa003b6d473a09d4271a7c46743ba8b6266a216c903bba5595bb84","side":"right"},{"sibling":"afa86de7d2b39f53748bea8c4274ea857aed4694d034da9da1628457bb87cbe4","side":"right"},{"sibling":"0adb35dd3ad9bfc4a781c02606acf0a3ce1e28e1a88066209cc1a35814e790ad","side":"right"},{"sibling":"f292d3278e493ec60902181b8c0bd5c89c0a1168ef928222fe6161287982f7f0","side":"right"},{"sibling":"2209295faf1a5bf51c97c6fd5a839a8181a4ea44f490420381530f35df7d9b2f","side":"right"},{"sibling":"5784576a15214ea9fc3569e6e1cff1ef443c0b1fc0d036028489088af089de27","side":"right"},{"sibling":"311772ec218efcb2da5a337f9e9f042fe1cc0028643adb0a354787e4ea7911b7","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"7b6f0bea4291a5e63574dfca9aa0f9756f450c9478c3d07474d39a7ababb51f9","side":"right"},{"sibling":"1d39fe14b21e2ebbfb87e882423b24ee9469eae1e4c77af5b799ac4db9537467","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":156177,"merkle_root":"c5705a0243d16afd8b1ebfd731b7aa304079c442c2a7906493c5bbed374c69ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260528T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-28T05:37:36Z","sig_algorithm":"ed25519","signature":"f087e13febc8bb6a2e0812610de64cebc65be92915518d9c4b230799c3b161839b04c4eb1b02741f938f35545a76ab76555b04c782bdc2f9a44852d171d65909","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b848631fc239498dfb6daf26aa8634d59b9d2331102ce0e6acc35a1348820ad2"}}