{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c6b79469d0cdb5d88c4e5190f17a4f6e0faa34578c88d5f38981ed3e477b9546","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c6b79469d0cdb5d88c4e5190f17a4f6e0faa34578c88d5f38981ed3e477b9546","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e08526f6f895a9f060fe462bdd52f058917bce85fa31643c79231a2df1bf5f40","published":"Thu, 18 Jun 2026 00:00:00 -0400","receipt_hash":"e08526f6f895a9f060fe462bdd52f058917bce85fa31643c79231a2df1bf5f40","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e08526f6f895a9f060fe462bdd52f058917bce85fa31643c79231a2df1bf5f40","observed_at":"2026-06-18T04:43:37.219665Z","parent_run_hash":"de79a40f7b3537d88842f7ac355e799c5df2adcb4fc32a4e28096d4bbdf01739","published":"Thu, 18 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2603.00656v2 Announce Type: replace \nAbstract: Real-world user requests to LLM agents are often underspecified. Agents must interact to acquire missing information and make correct downstream decisions. However, current multi-turn GRPO-based methods often rely on trajectory-level reward computation, which leads to credit assignment problems and insufficient advantage signals within rollout groups. A feasible approach is to identify valuable interaction turns at a fine granularity to drive more targeted learning. To address this, we introduce InfoPO (Information-Driven Policy Optimization), which frames multi-turn interaction as a process of active uncertainty reduction and computes an information-gain reward that credits turns whose feedback measurably changes the agent's subsequent action distribution compared to a masked-feedback counterfactual. It then combines this signal with task outcomes via an adaptive variance-gated fusion to identify information importance while maintai","title":"InfoPO: Information-Driven Policy Optimization for User-Centric Agents","url":"https://arxiv.org/abs/2603.00656","vendor":"arxiv_cs_ai"},"summary":"arXiv:2603.00656v2 Announce Type: replace \nAbstract: Real-world user requests to LLM agents are often underspecified. Agents must interact to acquire missing information and make correct downstream decisions. However, current multi-turn GRPO-based methods often rely on trajectory-level reward computation, which leads to credit assignment problems and insufficient advantage signals within rollout groups. A feasible approach is to identify valuable interaction turns at a fine granularity to drive more targeted learning. To address this, we introduce InfoPO (Information-Driven Policy Optimization), which frames multi-turn interaction as a process of active uncertainty reduction and computes an information-gain reward that credits turns whose feedback measurably changes the agent's subsequent action distribution compared to a masked-feedback counterfactual. It then combines this signal with task outcomes via an adaptive variance-gated fusion to identify information importance while maintai","title":"InfoPO: Information-Driven Policy Optimization for User-Centric Agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-18T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2603.00656"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:e9de49f8ca2b483af85f1ad3b3d56ff9af26cec7d0279a57882ce9a9c35a37df2baf6187dcb36d17290056277ae9d640f6bd3c6d338144de3bc15153e89fd209","signer":"crovia.substrate","subject":{"observed_at":"2026-06-18T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2603.00656"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8107944897c11b40302a458ce32c8bcfd7a0e39dee61f34db9fcc1cdfff37f40","leaf_index":233435,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"24ce8aa03784528f8b41f85cbfaf9c0a377b1b2a1e171c2477c60d8617ca6a9f","side":"left"},{"sibling":"a216630d1f4ec80bee739b31c4227de939a09b5a6e05ca4b57fe1d20ce8a0ea6","side":"left"},{"sibling":"4a81aa8dddf1e10ca0c034dce49eb749e02c3c04fd8a06d3982ea114c0a869ea","side":"right"},{"sibling":"d5062ad1ea4051e81a20554ca5f6ab50c988c216252df0a79e1b5e86d2b345e2","side":"left"},{"sibling":"e87d3586a31858fbf8b65081695b952d3a108457207fefae8bf8247f1d2a9f5e","side":"left"},{"sibling":"8c610da56e97c6514b09dd5f45c229dc3a880e5ee127aa6c1cb747aa9abcccc0","side":"right"},{"sibling":"8942c0022e58c4389bb72bd28400c56919b0d25c91e8d4c9f09dec03626b6f4b","side":"left"},{"sibling":"b4c26795680b2096400acbdc34159290b2a0589778819fc7e052c7a60bb5c873","side":"left"},{"sibling":"429c2a92a65e6eeaa2eda0a35fdb9e541472a1eace4c69a4d01a618a659a110f","side":"left"},{"sibling":"d97d1ebe04af6ea572f9d4334004d01026acffa3da5be2883c7566513c76e2c0","side":"left"},{"sibling":"571eb56e7ce00fe1f38d0ac4fc56828d01b2cfc1ab9089cde245c0656bee0514","side":"left"},{"sibling":"7ac50038a8ced3aeaf1194a2407a4a09346b0e4399da36ecde4390675a0c4bf1","side":"left"},{"sibling":"8c5e2b48dc31ef0edcd35c3db048235aa78cc48443aa2a8da3aa6e9b5524d2c4","side":"right"},{"sibling":"e616c34dbaf9456d5a6d3e2da82cde8621293c9f6d8a4cf9e441d7fd9cc81579","side":"right"},{"sibling":"94c0c932e61657f5e37fdba43f6ca9eddea8359425a7c1558dabe566911d5304","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":234491,"merkle_root":"02576a6980e38bab47864ae2c57b5a5ff21e554e9bdf8f64bdf28155ff1aabec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260618T143732Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-18T18:33:39Z","sig_algorithm":"ed25519","signature":"b6c708778fc38b7789a2b91156cfe87252a7cd3a1d29121cba11a0c78f8cf104ca3019fc50a962fa5a216bcc4922fc8f3f69c04d1f8332c6dc0d931e1012e502","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c6b79469d0cdb5d88c4e5190f17a4f6e0faa34578c88d5f38981ed3e477b9546"}}