{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_338445f8ef6f295689370908b2fc15a8314e5baf59a50599024a534e159affd1","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_338445f8ef6f295689370908b2fc15a8314e5baf59a50599024a534e159affd1","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"69ba8326fb74c02d7c0163dd98dc8b34709d01cd0b92d39d56ff083b7492ceb3","published":"Wed, 20 May 2026 00:00:00 -0400","receipt_hash":"69ba8326fb74c02d7c0163dd98dc8b34709d01cd0b92d39d56ff083b7492ceb3","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"69ba8326fb74c02d7c0163dd98dc8b34709d01cd0b92d39d56ff083b7492ceb3","observed_at":"2026-05-20T04:43:44.562035Z","parent_run_hash":"5f904c2c2fecb6b44f2adce8bdc9de914b9a39f7c4fdd7bf086a6c2361a30f8c","published":"Wed, 20 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.19425v1 Announce Type: cross \nAbstract: Reinforcement Learning with Verifiable Rewards (RLVR) has become the dominant paradigm for advanced reasoning in Large Language Models (LLMs), but rollout samples are expensive to obtain, making sample efficiency a critical bottleneck. A natural remedy is to reuse each rollout batch for multiple gradient updates, a standard practice in classical RL. Yet in RLVR, this amplifies policy shift, leading to severe performance degradation. Detecting the onset of degradation early enough to stop reuse remains an open and challenging problem. We close this gap by identifying the \\textit{Disproportionate Weight Divergence (DWD)} phenomenon: performance degradation is synchronized with a sharp surge in the \\texttt{lm\\_head} weight change, while intermediate layers remain stable. Empirically, we verify that DWD emerges consistently across diverse LLMs and tasks. Theoretically, we prove that (i) harmful gradients concentrate at the \\texttt{lm\\_head","title":"When to Stop Reusing: Dynamic Gradient Gating for Sample-Efficient RLVR","url":"https://arxiv.org/abs/2605.19425","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.19425v1 Announce Type: cross \nAbstract: Reinforcement Learning with Verifiable Rewards (RLVR) has become the dominant paradigm for advanced reasoning in Large Language Models (LLMs), but rollout samples are expensive to obtain, making sample efficiency a critical bottleneck. A natural remedy is to reuse each rollout batch for multiple gradient updates, a standard practice in classical RL. Yet in RLVR, this amplifies policy shift, leading to severe performance degradation. Detecting the onset of degradation early enough to stop reuse remains an open and challenging problem. We close this gap by identifying the \\textit{Disproportionate Weight Divergence (DWD)} phenomenon: performance degradation is synchronized with a sharp surge in the \\texttt{lm\\_head} weight change, while intermediate layers remain stable. Empirically, we verify that DWD emerges consistently across diverse LLMs and tasks. Theoretically, we prove that (i) harmful gradients concentrate at the \\texttt{lm\\_head","title":"When to Stop Reusing: Dynamic Gradient Gating for Sample-Efficient RLVR","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-20T04:43:44Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.19425"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:237603db14eebcfb033d5ccee1bb525c3b09b7428521d02c5f0777bf54b6094b27017a12998e66668730e841a64ce3dccb4cce54f4282aedf9f520122ddfd507","signer":"crovia.substrate","subject":{"observed_at":"2026-05-20T04:43:44Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.19425"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"afd10be3b76f0eee5fecbdeb2c345fefef20e2c34cb50e10da105b8195bc7045","leaf_index":145123,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"ca92605dd8cf376be27c4998f2bf416b824bdcaa541add0d7f4970e62a8b1c77","side":"left"},{"sibling":"9f339eb23d32c766085a2dd1061c313966762459962bdab2319fd943cdf6f009","side":"left"},{"sibling":"1f04f6b2b85b53563d61b01c900f5a92be1e4156c52f94fdf9ea2a0d8c8db31f","side":"right"},{"sibling":"caf79d0643b50dd6104bb06f4843480a1611d5b7b2fdd4b5e81e2a80ca51e527","side":"right"},{"sibling":"b97803c71d5215cee6366886f1c2bf238f5926efcfabe0f7998629b68e71a375","side":"right"},{"sibling":"c20d81eca743ab1f901caec96410a08534c42f0b1eeffc511625af41a4fa05af","side":"left"},{"sibling":"14b33559692f11ef077e3863ff56431a676a64de58614d8dda5f4a9e91838cf8","side":"left"},{"sibling":"3047d0cbdc103ff93a1a073ea940e12ff062595dae419513748ed63fbcb3359c","side":"left"},{"sibling":"9db04d2181fb869f16a5baf7dcfe999cc93a7e9beef8493d0cc71f57470fe5e1","side":"right"},{"sibling":"1526885f19d1fadf6955cf519dbc4e62d593a4bba99d741e8c301740a7068233","side":"left"},{"sibling":"e3a7d5c07f161682d61bd453ffc02ecdf87cfeda70f986d6650017f9d2d6b265","side":"left"},{"sibling":"edbc49f08e5b92291934c05c9e6efd270a6b0698d8d2fa474006366027dfe098","side":"right"},{"sibling":"3e4df6e7457cecbf36f350375e72dcab336a3984422e4c406ef809e4e2944e96","side":"left"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"be08fedc4e72a6fb56606f66812fae7317e09690b9acb18385f4ab117a981238","side":"right"},{"sibling":"0534329a7475dc9df51998c83c16892126679dade0fa34182f21e869599386c7","side":"right"},{"sibling":"2d24720928ead0e7670650eb55f558c4f20e4c18df376f47ba72cfa8cf0ed344","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":147301,"merkle_root":"08903d7159c3b38eeeeafc09eab15139ea417f1d94f02f1fbc87296b37db840a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260521T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-21T18:37:33Z","sig_algorithm":"ed25519","signature":"905f2924632dfa2970c8690285f5b5d4a1d891d0e0ef1cbc404ebec2fd937215ac768e16f0a9f28b18977a55ae0bfd226db6833ae7ef588729054117d2da7303","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_338445f8ef6f295689370908b2fc15a8314e5baf59a50599024a534e159affd1"}}