{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_a9abf7b7087b56d181d42571d7ce125cc33c0b06bf6c52fd30ec9f036a7e997f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_a9abf7b7087b56d181d42571d7ce125cc33c0b06bf6c52fd30ec9f036a7e997f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"bf82249606acfde6d2c1824b3151a4e0ab0e30874acd9d5fb52048f7af025985","published":"Tue, 21 Jul 2026 00:00:00 -0400","receipt_hash":"bf82249606acfde6d2c1824b3151a4e0ab0e30874acd9d5fb52048f7af025985","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"bf82249606acfde6d2c1824b3151a4e0ab0e30874acd9d5fb52048f7af025985","observed_at":"2026-07-21T04:43:35.036805Z","parent_run_hash":"03e944014de2697434479833d15ea9303e014945afc230ecc7f207824493b589","published":"Tue, 21 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2508.04225v4 Announce Type: replace-cross \nAbstract: Behavior Regularized Policy Optimization (BRPO) leverages asymmetric divergence regularization to mitigate distribution shift in offline reinforcement learning. This paper is the first to study the open question of symmetric BRPO. Using didactic examples, we show that symmetric regularization can outperform asymmetric regularization in addressing one-sided bias, near-boundary policy updates, and projection geometry consistency. However, symmetric divergences do not fit BRPO naturally: they do not permit a closed-form solution when used as regularizers, and can lead to numerical instability when used as optimization objectives. We first introduce a universal BRPO framework using an infinite series of Pearson-Vajda divergences to represent any $f$-divergence, which includes both symmetric and asymmetric divergences. We use a finite-series approximation to obtain the following results for symmetric BRPO: (1) a closed-form optimal ","title":"Symmetric Behavior Regularized Policy Optimization","url":"https://arxiv.org/abs/2508.04225","vendor":"arxiv_cs_ai"},"summary":"arXiv:2508.04225v4 Announce Type: replace-cross \nAbstract: Behavior Regularized Policy Optimization (BRPO) leverages asymmetric divergence regularization to mitigate distribution shift in offline reinforcement learning. This paper is the first to study the open question of symmetric BRPO. Using didactic examples, we show that symmetric regularization can outperform asymmetric regularization in addressing one-sided bias, near-boundary policy updates, and projection geometry consistency. However, symmetric divergences do not fit BRPO naturally: they do not permit a closed-form solution when used as regularizers, and can lead to numerical instability when used as optimization objectives. We first introduce a universal BRPO framework using an infinite series of Pearson-Vajda divergences to represent any $f$-divergence, which includes both symmetric and asymmetric divergences. We use a finite-series approximation to obtain the following results for symmetric BRPO: (1) a closed-form optimal ","title":"Symmetric Behavior Regularized Policy Optimization","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-21T04:43:35Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2508.04225"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:26e4965eecaf71e9cd09f10809190aeb7811cbb70f16cf02f1a01093214b602f5994f7f88fb22d3211d915e0debf7ad5e4328d8c45602d91d9ecc3f57c10940d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-21T04:43:35Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2508.04225"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"564db281707f66c94d6bf10b840936806d5af7aed77b5414baf437410e79b2df","leaf_index":336920,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"0208e1793175d4701f34c572a8c7a3b2380cb4e979e38a728fa3f60ac89cb3a0","side":"right"},{"sibling":"205a43a8ab0267df77cf8fc9729f0bf41c344f4ddda532469fac6d9694422be3","side":"right"},{"sibling":"b4d7e0238b2c1c971b69a6a69c9a921002fc94937919fe555efd2005197b6bd2","side":"right"},{"sibling":"d89012ad4b4b04af4b5b85e5621247729e7e9285d4e581399129044abafec755","side":"left"},{"sibling":"7d21e41d8dbce978fcdadd7bfac6afaf79d14d4b14efd41f4eb6d9a6d02090d9","side":"left"},{"sibling":"970de1a9ae0755f91297bfb952a2e1b5408d3c6ffa1792b61545c081aaf5382d","side":"right"},{"sibling":"950013dbd7f9f5861e4c5ff582a4f28f41529fee5ece541adb6d424c9cbd1f75","side":"right"},{"sibling":"45255096c5a08e740a615b6e3c262589a170dbfc373bdeaceae175916dd2ab97","side":"right"},{"sibling":"1298242aa509bc13b92160250ead2d95a721941ca6310e33b9eb0bf0821557d0","side":"right"},{"sibling":"61d44c6d16b87de577db53cff0dba2f003c50c036d92dc335120aa70124c8788","side":"right"},{"sibling":"5d0b792dbe69fd0a7cd8d79b467aa10204d1f25be602c06190f3423bc22727b8","side":"left"},{"sibling":"9eb5077edfb3dc553857d4794b925bfce117e0f8a1d049af5d0dd9026b470eef","side":"right"},{"sibling":"414b1a70fd1dcb25489a194714b97492b066684b15d0b7a48a176c4b9b5bc713","side":"right"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"613f015699131eb89bd755dee67133be95af25cf5f16c1c8ce4b99d963b8dd86","side":"right"},{"sibling":"a729b574b1135956436ded5eef1fe8f08014ff6a0729749d307ab1bca93fcdc9","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"4bf21052e085e8ac81f1dec1d2b310bd12bf948992de6177d12e9d2fda8d39f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":337144,"merkle_root":"5e969cc01afa67e4dbe5d37b712cdb10f4aa1fd74404e02eab724cf487c8d6d9","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260721T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-21T05:38:38Z","sig_algorithm":"ed25519","signature":"5c0c1a8dd2793d787ccd5e49e8b4d70eed136352555518589f05c74be357fc171702e42a76c3a556d90e3d51ff36cb3d292aaac83c66566de7b943f318bda50c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_a9abf7b7087b56d181d42571d7ce125cc33c0b06bf6c52fd30ec9f036a7e997f"}}