{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c74d34eb416dce5c2b2d9c4ece7e8a71067e7e5cfd095e135c1672c80ee5f044","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c74d34eb416dce5c2b2d9c4ece7e8a71067e7e5cfd095e135c1672c80ee5f044","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"bc9f996a877932e9706712850ec51cc40579d36ceec1d5b06092ceaa5af93b23","published":"Wed, 22 Jul 2026 00:00:00 -0400","receipt_hash":"bc9f996a877932e9706712850ec51cc40579d36ceec1d5b06092ceaa5af93b23","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"bc9f996a877932e9706712850ec51cc40579d36ceec1d5b06092ceaa5af93b23","observed_at":"2026-07-22T04:43:18.261256Z","parent_run_hash":"4765c85b8b4b27ff9a690c1ae11c3b009baa2c60297295f395ad422f5afed68c","published":"Wed, 22 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.10481v2 Announce Type: replace-cross \nAbstract: Reinforcement learning (RL) has significantly enhanced the reasoning capabilities of large language models (LLMs), yet the training process remains notoriously fragile. In this work, we investigate a critical source of this instability: over-optimization, where models exploit training heuristics at the expense of generalizable reasoning. While reverse KL regularization is the standard defense against such degradation, our analysis reveals that it is often insufficient in this regime, as it fails to ensure comprehensive coverage of the reference distribution. To address this, we propose ARMOR (Anchor Rollout and Mixed Optimization for RL), a framework that shifts the paradigm from passive penalty to active sample stabilization. ARMOR comprises two key components: (1) Anchor Rollout, which leverages off-policy data from the reference policy to preserve established solution patterns; and (2) Mixed Optimization, which reformulates ","title":"ARMOR: Stabilizing On-Policy LLM RL with Off-Policy Anchor Samples","url":"https://arxiv.org/abs/2607.10481","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.10481v2 Announce Type: replace-cross \nAbstract: Reinforcement learning (RL) has significantly enhanced the reasoning capabilities of large language models (LLMs), yet the training process remains notoriously fragile. In this work, we investigate a critical source of this instability: over-optimization, where models exploit training heuristics at the expense of generalizable reasoning. While reverse KL regularization is the standard defense against such degradation, our analysis reveals that it is often insufficient in this regime, as it fails to ensure comprehensive coverage of the reference distribution. To address this, we propose ARMOR (Anchor Rollout and Mixed Optimization for RL), a framework that shifts the paradigm from passive penalty to active sample stabilization. ARMOR comprises two key components: (1) Anchor Rollout, which leverages off-policy data from the reference policy to preserve established solution patterns; and (2) Mixed Optimization, which reformulates ","title":"ARMOR: Stabilizing On-Policy LLM RL with Off-Policy Anchor Samples","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-22T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.10481"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:bad4ec556214bf9f2b4215f4f88532650663e7067d8191257ec62e5fc6398c8bbac56a47d03d7a66c2551aeec4fce4de1a0115b2f2618d49ac0a48e8806a090c","signer":"crovia.substrate","subject":{"observed_at":"2026-07-22T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.10481"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"02eb56b9aa321076b1eed9123b76dc0677454abb244bbfd7716c891feafbc936","leaf_index":340426,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"a8af32133b84f5dae0ef0a6fad6876dc37d6b3751e7859daff0390dba16dbf89","side":"right"},{"sibling":"bdf7e3ed79588f9d2e265ab8e76394633ddca90ad4381653b27168235ecfd647","side":"left"},{"sibling":"f07a1c110b7fe3903e42b01f1e6a3f0cd6dd0afbd7513f6f7a48859028a471be","side":"right"},{"sibling":"2c7a35af507fe5c8733f32f18562b8b33a070dc508dc7c5a883533df76315609","side":"left"},{"sibling":"19b2ed8d087fedaee101559083db0b80d75c923d902b218d89860e6af335aae4","side":"right"},{"sibling":"a57e92d63b010c5b62c98864bb8cae1dcb58e4fc847798938d37f421f42b19d2","side":"right"},{"sibling":"59ec0f16698aa66a7cbc22ab77872ce579117827bc4c3ee9e52b0edfef8840fc","side":"left"},{"sibling":"5666268df88c3cf54274085bda986b69261da088d4bea21e6975025aadd566d3","side":"left"},{"sibling":"99d4b7aea5e7c917c580af0f1a9556bbd4d44f3f36cf9391892e7ef0340a8258","side":"left"},{"sibling":"c7fc9d4187cdc36f4c03b4b13daf4b880ea65536b051f71a5cc2543839d02697","side":"right"},{"sibling":"1758ec6ac206ce40e8368cb702195322fe3737d0fb03d8bd9e3b30acc4fa7d81","side":"right"},{"sibling":"0c407f0d553cf3fab8f9bd79205b8180e090cbf29fa0490ebb55155041ad5c86","side":"right"},{"sibling":"2dd9cb2521044ee7c6b74f2315e0a0253b8df0d04a7b810bbbbe7da5a9788769","side":"left"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"787ee3744642ff909d610b0514cb100784f0ef1ba0ef4c70a0dc91f0ab2bb192","side":"right"},{"sibling":"9fc8a8ebbc1bff7e62b9f1e1c681c91e7196092ce9551573df6e23096df13e4d","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"ee6f33920899d9bdef2eb706dfff29e26eea61e29ea9824c6c8e6bcd48275d76","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":340557,"merkle_root":"7d45d94f20b5bf82263df45b87749e19e06161972f25141dd573cc138d566338","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260722T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-22T05:38:39Z","sig_algorithm":"ed25519","signature":"812cb61e90d3582ba508db8515c4168bbf8ee1c6762885609049d012f680f8068f15c98b9accf2042047dcfc6a6fc3885820ee34b0be350846386f64f2591309","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c74d34eb416dce5c2b2d9c4ece7e8a71067e7e5cfd095e135c1672c80ee5f044"}}