{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_adde6530f7501cf4de3385cee5bc51b0dfa0b4010d9d5d6c2beb6d69d930c4b4","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_adde6530f7501cf4de3385cee5bc51b0dfa0b4010d9d5d6c2beb6d69d930c4b4","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"3cbc16380bf314e231a173d6809941321b3d812f7a1a3e5c768095df3a147ce0","published":"Tue, 16 Jun 2026 00:00:00 -0400","receipt_hash":"3cbc16380bf314e231a173d6809941321b3d812f7a1a3e5c768095df3a147ce0","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"3cbc16380bf314e231a173d6809941321b3d812f7a1a3e5c768095df3a147ce0","observed_at":"2026-06-16T04:43:43.281320Z","parent_run_hash":"eb6edcf82c3507c59161a4ab46d2e904e507004f44677402bb24d106997ed7c2","published":"Tue, 16 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2510.01444v3 Announce Type: replace \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has advanced reasoning capabilities in multimodal large language models. However, existing methods typically treat visual inputs as deterministic, overlooking the perceptual ambiguity inherent to the visual modality. Consequently, they fail to distinguish whether a model's uncertainty stems from complex reasoning or ambiguous perception, preventing the targeted allocation of exploration or learning signals. To address this gap, we introduce \\textbf{DUPL}, a dual-uncertainty guided policy learning approach for multimodal RLVR that quantifies and leverages both perceptual uncertainty (via symmetric KL divergence) and output uncertainty (via policy entropy) to guide policy updates. By establishing an uncertainty-driven feedback loop and employing a dynamic branch prioritization mechanism, DUPL recalibrates the policy advantage to focus learning on states with high perceptual or decis","title":"Dual-Uncertainty Guided Policy Learning for Multimodal Reasoning","url":"https://arxiv.org/abs/2510.01444","vendor":"arxiv_cs_ai"},"summary":"arXiv:2510.01444v3 Announce Type: replace \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has advanced reasoning capabilities in multimodal large language models. However, existing methods typically treat visual inputs as deterministic, overlooking the perceptual ambiguity inherent to the visual modality. Consequently, they fail to distinguish whether a model's uncertainty stems from complex reasoning or ambiguous perception, preventing the targeted allocation of exploration or learning signals. To address this gap, we introduce \\textbf{DUPL}, a dual-uncertainty guided policy learning approach for multimodal RLVR that quantifies and leverages both perceptual uncertainty (via symmetric KL divergence) and output uncertainty (via policy entropy) to guide policy updates. By establishing an uncertainty-driven feedback loop and employing a dynamic branch prioritization mechanism, DUPL recalibrates the policy advantage to focus learning on states with high perceptual or decis","title":"Dual-Uncertainty Guided Policy Learning for Multimodal Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-16T04:43:43Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2510.01444"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4b78b5fe535294c81e664ebdfb9fa738624e09be2c83265c4d589005bc722582dd4836b8ea3e681439739f768a39153ece503fed220f883117e378cc17973b0f","signer":"crovia.substrate","subject":{"observed_at":"2026-06-16T04:43:43Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2510.01444"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"468de5f667e8b061ac67e829a6c949f391220534a17eee84c3e4e01fb04c406c","leaf_index":230622,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"11232b24aff85e4b958b5afe7582a3870c88feb2c3acecf259621f41b300ac06","side":"right"},{"sibling":"a77f020c2b2e97e29ec5aad6b2481f076632fe6abc37132f8b07f63bb0096c6f","side":"left"},{"sibling":"0abba0ad59fcaab5ea3e748d59f0f2ede7444f55be714fdd3f88fbdd588424d2","side":"left"},{"sibling":"d386b5ea0fbaa7a47960c3ec04e82b3836f567d6e10c764869552d83420e9765","side":"left"},{"sibling":"f5e89b71b08d06f91991bfd9a01d835895d8f11c6c49694858aa6a356c942174","side":"left"},{"sibling":"b23a457f9b965a633c26bab66026ad769d97ec466ee70b0741e924a50578df05","side":"right"},{"sibling":"f22cc217fca41b12942eb7a1a3f3a258a96e32aee3539815798f453610b93fdc","side":"left"},{"sibling":"f5958932b707fbb8d9de8cb158fe61200709928e9e5d860d183ba06f040858b6","side":"left"},{"sibling":"63d6b9d8af7ae8348285d3493af29f64992ecec42f5f1cae99c8604cd1703487","side":"right"},{"sibling":"0c5669692381d605223c74b8d30f70cd308e77e33d5e40ea84bb7b4f84f2d4d9","side":"right"},{"sibling":"d5b9f8b1a2c9f6a46e17982dfbe6ce1f3b5fa4e730220397f2253d114dcc8486","side":"left"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_adde6530f7501cf4de3385cee5bc51b0dfa0b4010d9d5d6c2beb6d69d930c4b4"}}