{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_03d754ea95e5101dc42cbfeee83e6f53e4a163459bbf07df16657802b58fa759","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_03d754ea95e5101dc42cbfeee83e6f53e4a163459bbf07df16657802b58fa759","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"65e2186850a378bc2a2876a59b18c42edb51292d800c27be33fc8f316b144921","published":"Thu, 18 Jun 2026 00:00:00 -0400","receipt_hash":"65e2186850a378bc2a2876a59b18c42edb51292d800c27be33fc8f316b144921","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"65e2186850a378bc2a2876a59b18c42edb51292d800c27be33fc8f316b144921","observed_at":"2026-06-18T04:43:37.219665Z","parent_run_hash":"de79a40f7b3537d88842f7ac355e799c5df2adcb4fc32a4e28096d4bbdf01739","published":"Thu, 18 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2603.09344v3 Announce Type: replace \nAbstract: Offline reinforcement learning (RL) enables data-efficient and safe policy learning without online exploration, but its performance often degrades under distribution shift. The learned policy may visit out-of-distribution state-action pairs where value estimates and learned dynamics are unreliable. To address policy-induced extrapolation and transition uncertainty in a unified framework, we formulate offline RL as robust policy optimization, treating the transition kernel as a decision variable within an uncertainty set and optimizing the policy against the worst-case dynamics. We propose Robust Regularized Policy Iteration (RRPI), which replaces the intractable max-min bilevel objective with a tractable KL-regularized surrogate and derives an efficient policy iteration procedure based on a robust regularized Bellman operator. We provide theoretical guarantees by showing that the proposed operator is a $\\gamma$-contraction and that i","title":"Robust Regularized Policy Iteration under Transition Uncertainty","url":"https://arxiv.org/abs/2603.09344","vendor":"arxiv_cs_ai"},"summary":"arXiv:2603.09344v3 Announce Type: replace \nAbstract: Offline reinforcement learning (RL) enables data-efficient and safe policy learning without online exploration, but its performance often degrades under distribution shift. The learned policy may visit out-of-distribution state-action pairs where value estimates and learned dynamics are unreliable. To address policy-induced extrapolation and transition uncertainty in a unified framework, we formulate offline RL as robust policy optimization, treating the transition kernel as a decision variable within an uncertainty set and optimizing the policy against the worst-case dynamics. We propose Robust Regularized Policy Iteration (RRPI), which replaces the intractable max-min bilevel objective with a tractable KL-regularized surrogate and derives an efficient policy iteration procedure based on a robust regularized Bellman operator. We provide theoretical guarantees by showing that the proposed operator is a $\\gamma$-contraction and that i","title":"Robust Regularized Policy Iteration under Transition Uncertainty","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-18T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2603.09344"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:82f4d6ad9deb929b954f5fa7c904f2af118ece87b1c13673e725de62580fa8388c69d402cb6283d1ca1d5ebf337208a966180e5bb03a981cd8c73602663ba905","signer":"crovia.substrate","subject":{"observed_at":"2026-06-18T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2603.09344"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1d3850fd07028ffde95c414b95a874e14455856f21135804a0b1176b4b612690","leaf_index":233436,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"937fe4694560c0d0bc23de37ffaf6290baff652e191e89d6d3f05894ec563a2e","side":"right"},{"sibling":"23940b3ee96cc49bde8c87c9f9ceee73ae3b75233e205a83bbe9af3bfa45ce08","side":"right"},{"sibling":"d687a8efeede1a6c6d61a2f97ab9b2fb7bbebf969ab4cfb86dc7cda814df96c5","side":"left"},{"sibling":"d5062ad1ea4051e81a20554ca5f6ab50c988c216252df0a79e1b5e86d2b345e2","side":"left"},{"sibling":"e87d3586a31858fbf8b65081695b952d3a108457207fefae8bf8247f1d2a9f5e","side":"left"},{"sibling":"8c610da56e97c6514b09dd5f45c229dc3a880e5ee127aa6c1cb747aa9abcccc0","side":"right"},{"sibling":"8942c0022e58c4389bb72bd28400c56919b0d25c91e8d4c9f09dec03626b6f4b","side":"left"},{"sibling":"b4c26795680b2096400acbdc34159290b2a0589778819fc7e052c7a60bb5c873","side":"left"},{"sibling":"429c2a92a65e6eeaa2eda0a35fdb9e541472a1eace4c69a4d01a618a659a110f","side":"left"},{"sibling":"d97d1ebe04af6ea572f9d4334004d01026acffa3da5be2883c7566513c76e2c0","side":"left"},{"sibling":"571eb56e7ce00fe1f38d0ac4fc56828d01b2cfc1ab9089cde245c0656bee0514","side":"left"},{"sibling":"7ac50038a8ced3aeaf1194a2407a4a09346b0e4399da36ecde4390675a0c4bf1","side":"left"},{"sibling":"8c5e2b48dc31ef0edcd35c3db048235aa78cc48443aa2a8da3aa6e9b5524d2c4","side":"right"},{"sibling":"e616c34dbaf9456d5a6d3e2da82cde8621293c9f6d8a4cf9e441d7fd9cc81579","side":"right"},{"sibling":"94c0c932e61657f5e37fdba43f6ca9eddea8359425a7c1558dabe566911d5304","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":234491,"merkle_root":"02576a6980e38bab47864ae2c57b5a5ff21e554e9bdf8f64bdf28155ff1aabec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260618T143732Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-18T18:33:39Z","sig_algorithm":"ed25519","signature":"b6c708778fc38b7789a2b91156cfe87252a7cd3a1d29121cba11a0c78f8cf104ca3019fc50a962fa5a216bcc4922fc8f3f69c04d1f8332c6dc0d931e1012e502","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_03d754ea95e5101dc42cbfeee83e6f53e4a163459bbf07df16657802b58fa759"}}