{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_34d07ce86e1353d5dbc82596edec4fd4c7ddf3003f3c44423f0bb228526a91ee","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_34d07ce86e1353d5dbc82596edec4fd4c7ddf3003f3c44423f0bb228526a91ee","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"dece4b5560fa404aa130fc05a8cbebfee43b17c291fc110aa1b803e7f109077c","published":"Tue, 21 Jul 2026 00:00:00 -0400","receipt_hash":"dece4b5560fa404aa130fc05a8cbebfee43b17c291fc110aa1b803e7f109077c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"dece4b5560fa404aa130fc05a8cbebfee43b17c291fc110aa1b803e7f109077c","observed_at":"2026-07-21T04:43:35.036805Z","parent_run_hash":"03e944014de2697434479833d15ea9303e014945afc230ecc7f207824493b589","published":"Tue, 21 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.00680v3 Announce Type: replace \nAbstract: Offline reinforcement learning (RL) aims to optimize policies from pre-collected datasets. A bottleneck of this paradigm is managing epistemic uncertainty, which arises from limited data coverage (sample-level) and the ambiguity in identifying transition dynamics from finite data (model-level). To provide a unified quantification of these uncertainties, Bayesian RL has been proposed by treating the dynamics model as a random variable and maintaining a corresponding belief. Despite its theoretical appeal, policy optimization in Bayesian RL remains computationally challenging as it requires solving composite objectives with expectations. Prior methods either employ search-based techniques with poor computational scalability or impose restrictive posterior assumptions that sacrifice the adaptability of Bayesian RL. To address these limitations, we propose Posterior Hybrid Bayesian Belief (PhyB), which reformulates the expectation as a c","title":"Regularized Offline Policy Optimization with Posterior Hybrid Bayesian Belief","url":"https://arxiv.org/abs/2606.00680","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.00680v3 Announce Type: replace \nAbstract: Offline reinforcement learning (RL) aims to optimize policies from pre-collected datasets. A bottleneck of this paradigm is managing epistemic uncertainty, which arises from limited data coverage (sample-level) and the ambiguity in identifying transition dynamics from finite data (model-level). To provide a unified quantification of these uncertainties, Bayesian RL has been proposed by treating the dynamics model as a random variable and maintaining a corresponding belief. Despite its theoretical appeal, policy optimization in Bayesian RL remains computationally challenging as it requires solving composite objectives with expectations. Prior methods either employ search-based techniques with poor computational scalability or impose restrictive posterior assumptions that sacrifice the adaptability of Bayesian RL. To address these limitations, we propose Posterior Hybrid Bayesian Belief (PhyB), which reformulates the expectation as a c","title":"Regularized Offline Policy Optimization with Posterior Hybrid Bayesian Belief","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-21T04:43:35Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.00680"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:112b324235a72f97a9e668654de731526577462397e98b5185771df84624a09a6935bad3698100ffaa28c5114b3ea7ad40e6479564d672650daf17dc4f238509","signer":"crovia.substrate","subject":{"observed_at":"2026-07-21T04:43:35Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.00680"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"e0cceae4b7385723279d42f5b1c2f9dd0ec7d9468d493d302442c54bddea95fc","leaf_index":336884,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"25df307ac7bdc771d752bc1953946bc410910a02c4dbe7711154be9e716aed58","side":"right"},{"sibling":"cbcc7417e000536761a64bb1c0bfc339aefa7d696ef3bf11788be25c1c7ab469","side":"right"},{"sibling":"3fcd848c5eea9e8def0a17bc42f3516b046c01fa025712d2e8a560b2db63395d","side":"left"},{"sibling":"9865fce60b002e713e01f8015810cba907f2000c5438f9c7a4680189b2799941","side":"right"},{"sibling":"5aea04dc959084dd2eea348e12c742bc20e46082d70768efa05d25b3e8be0ea4","side":"left"},{"sibling":"a65f38bde1a7766840f31e782bd927bf0dbda3bf3127a35ed41f906e4c88dbd5","side":"left"},{"sibling":"ce03a7473ecb89a8a8b71de92b9d2d6aa1aa165b6c59cefc86a37241b7f55b64","side":"left"},{"sibling":"90ee1b815719c44a83e23429473a176437135319947bc07c6e3752315a40068d","side":"left"},{"sibling":"bd233cee8c0876447824d86de33608c36e3a8e17dca0737155291ed768cf56d1","side":"left"},{"sibling":"7b927551b5db06b6571913b4e6792ffcce5291a3eca5a0df4a3b6e296271105f","side":"left"},{"sibling":"b77a0b5ae4607c8fe6ba73449d46b35076e3dedc0c82a2c65a05780d42a7bc2e","side":"right"},{"sibling":"9eb5077edfb3dc553857d4794b925bfce117e0f8a1d049af5d0dd9026b470eef","side":"right"},{"sibling":"414b1a70fd1dcb25489a194714b97492b066684b15d0b7a48a176c4b9b5bc713","side":"right"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"613f015699131eb89bd755dee67133be95af25cf5f16c1c8ce4b99d963b8dd86","side":"right"},{"sibling":"a729b574b1135956436ded5eef1fe8f08014ff6a0729749d307ab1bca93fcdc9","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"4bf21052e085e8ac81f1dec1d2b310bd12bf948992de6177d12e9d2fda8d39f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":337144,"merkle_root":"5e969cc01afa67e4dbe5d37b712cdb10f4aa1fd74404e02eab724cf487c8d6d9","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260721T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-21T05:38:38Z","sig_algorithm":"ed25519","signature":"5c0c1a8dd2793d787ccd5e49e8b4d70eed136352555518589f05c74be357fc171702e42a76c3a556d90e3d51ff36cb3d292aaac83c66566de7b943f318bda50c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_34d07ce86e1353d5dbc82596edec4fd4c7ddf3003f3c44423f0bb228526a91ee"}}