{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_89d8ddf1968068dbcb454384481a5df55fe9b4c7d69cd8cac1f0035e5d13d736","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_89d8ddf1968068dbcb454384481a5df55fe9b4c7d69cd8cac1f0035e5d13d736","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"4132601bbe51751e2032fb1b95f08df9c42c87d972dec12e1558d6bee8ba78da","published":"Wed, 10 Jun 2026 00:00:00 -0400","receipt_hash":"4132601bbe51751e2032fb1b95f08df9c42c87d972dec12e1558d6bee8ba78da","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"4132601bbe51751e2032fb1b95f08df9c42c87d972dec12e1558d6bee8ba78da","observed_at":"2026-06-10T04:43:37.461885Z","parent_run_hash":"23aff1a6f676ba7ca33f70f4ddfae1dd282fb86104d577ce9be510d81a94c5dc","published":"Wed, 10 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.10580v1 Announce Type: cross \nAbstract: The asymptotic behaviour of Monte Carlo optimistic policy iteration (MC-O-PI) is a long-standing open question. When the model of the environment is unknown, as is common in practice, the only known condition that guarantees convergence to optimality is impractical. In its canonical form, this condition requires that the episodes used for policy evaluation be initialised uniformly over the entire state-action space. This paper strictly relaxes that requirement. Specifically, we prove that initial-visit MC-O-PI converges to optimality even when updates are uniform only over the actions within each state. This allows episodes to start in different states at arbitrary frequencies; a realistic implementation when the state space is large or unknown but the action space in each state is manageable. The proof departs from the classical analysis of Tsitsiklis whose central commutativity argument no longer applies when states are updated at di","title":"Convergence of Monte Carlo Optimistic Policy Iteration: Beyond Uniform State-Action Updates","url":"https://arxiv.org/abs/2606.10580","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.10580v1 Announce Type: cross \nAbstract: The asymptotic behaviour of Monte Carlo optimistic policy iteration (MC-O-PI) is a long-standing open question. When the model of the environment is unknown, as is common in practice, the only known condition that guarantees convergence to optimality is impractical. In its canonical form, this condition requires that the episodes used for policy evaluation be initialised uniformly over the entire state-action space. This paper strictly relaxes that requirement. Specifically, we prove that initial-visit MC-O-PI converges to optimality even when updates are uniform only over the actions within each state. This allows episodes to start in different states at arbitrary frequencies; a realistic implementation when the state space is large or unknown but the action space in each state is manageable. The proof departs from the classical analysis of Tsitsiklis whose central commutativity argument no longer applies when states are updated at di","title":"Convergence of Monte Carlo Optimistic Policy Iteration: Beyond Uniform State-Action Updates","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-10T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.10580"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1fdb4e24752f7ecb113e80351b5f0161c665a8cf8e17647e0f61727c9325a54ef44de8469d788f3fc1f65778112a6ef6c954d35fc605dc3cdd6f652bb354a804","signer":"crovia.substrate","subject":{"observed_at":"2026-06-10T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.10580"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"83f5a8896e82d449a813a77a8116a66930cdb088530e3d3f60b4f83c5802c1a8","leaf_index":226049,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1d0f325259be25a55621e7d80d02106487b7d1222b328f17221b65feee31ac12","side":"left"},{"sibling":"8340a4900ca493b98367406628bf326982537fd705c9b08e5ee62f80e8793951","side":"right"},{"sibling":"7cc3138e24514b496281ff21d3d573431ef8c607f27357b76dab02f3279e6df5","side":"right"},{"sibling":"6de155ec8079f48b21f3ae4bb240876f52060d01c90f5868c8b99c500ea94444","side":"right"},{"sibling":"cb8401c9dec71a7621ca929a13c10319b3fca130d4a4f7f088f7a736d4a78c8a","side":"right"},{"sibling":"eff504b452805a63fe588ff0b83c952a08eb97faebb969a9d37a1ae8f1ecca0c","side":"right"},{"sibling":"88b2be21cb27d1f77926cea4c04fb2bafb96cdf167c1655dc8560cea37cb31bf","side":"right"},{"sibling":"670eed753a274b9b7494adeace1b88e7adfaecd7eeaa78b82e0fc83a75d20865","side":"right"},{"sibling":"b180cfc3f8638c912e90139ec42e2e3fd9b67b3fe342b36e8954314633ae5f61","side":"left"},{"sibling":"a4d17aefe58175050dc159af6246658fcf1c9f3ed57aacf1b350fc3261de4e69","side":"left"},{"sibling":"280b980aa0c7756b0b0cb22658f26466d36f0e70fbc3312cd2311d9898e30b8f","side":"right"},{"sibling":"c98954d4b658b1dda60fe52576fcf9bf21a2d49c67fb63f8c30f16ab5f721938","side":"right"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_89d8ddf1968068dbcb454384481a5df55fe9b4c7d69cd8cac1f0035e5d13d736"}}