{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1a70107754fc158289522e775ead02426bd35caa67e6fc33663520657a5f6580","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1a70107754fc158289522e775ead02426bd35caa67e6fc33663520657a5f6580","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"94a4a764d997918b651cce3bba9dcbe5ba9b7ff5cddcfaecdeaa356083625902","published":"Tue, 26 May 2026 00:00:00 -0400","receipt_hash":"94a4a764d997918b651cce3bba9dcbe5ba9b7ff5cddcfaecdeaa356083625902","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"94a4a764d997918b651cce3bba9dcbe5ba9b7ff5cddcfaecdeaa356083625902","observed_at":"2026-05-26T04:43:39.018238Z","parent_run_hash":"dca8dedd754ad6a1772113d6b97ee4f4ab9a0afbdeb44aaace5ff2d2446b164b","published":"Tue, 26 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.15620v5 Announce Type: replace-cross \nAbstract: Reinforcement Learning (RL) has significantly improved large language model reasoning, but existing RL fine-tuning methods rely heavily on heuristic techniques such as entropy regularization and reweighting to maintain stability. In practice, they often suffer from late-stage performance collapse, leading to degraded reasoning quality and unstable training. We identify a key factor behind this instability: a small fraction of tokens, termed spurious tokens (around 0.01%), which contribute little to the reasoning outcome but receive disproportionately amplified gradient updates due to inheriting the full sequence-level reward. We present a unified framework for evaluating token-level optimization impacts across spurious risk, gradient norms, and entropy changes. Building on the analysis of token characteristics that severely disrupt optimization, we propose the Silencing Spurious Tokens (S2T) mechanism to efficiently suppress th","title":"STAPO: Stabilizing Reinforcement Learning for LLMs by Silencing Rare Spurious Tokens","url":"https://arxiv.org/abs/2602.15620","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.15620v5 Announce Type: replace-cross \nAbstract: Reinforcement Learning (RL) has significantly improved large language model reasoning, but existing RL fine-tuning methods rely heavily on heuristic techniques such as entropy regularization and reweighting to maintain stability. In practice, they often suffer from late-stage performance collapse, leading to degraded reasoning quality and unstable training. We identify a key factor behind this instability: a small fraction of tokens, termed spurious tokens (around 0.01%), which contribute little to the reasoning outcome but receive disproportionately amplified gradient updates due to inheriting the full sequence-level reward. We present a unified framework for evaluating token-level optimization impacts across spurious risk, gradient norms, and entropy changes. Building on the analysis of token characteristics that severely disrupt optimization, we propose the Silencing Spurious Tokens (S2T) mechanism to efficiently suppress th","title":"STAPO: Stabilizing Reinforcement Learning for LLMs by Silencing Rare Spurious Tokens","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-26T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.15620"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1b392e352ad3ada61c583b6dec8363c7f67a5a63ac405c5754b695d3f8286e077f0c2da473b6b5f14ed4ad40f51a0726d4101cafb5b5cf008958d75f1be4920a","signer":"crovia.substrate","subject":{"observed_at":"2026-05-26T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.15620"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"6ed35fea383ee452f3395aaa59a2f5a2cb3de4621a96383896be5c7d79338050","leaf_index":151953,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c7db858e414eb9e598d73d8ae12516cf48d0e68a194dc2b147233cfafb1981a0","side":"left"},{"sibling":"a8f71278324c82cc601d068982fafa1d3185f8672f5605bff1e6944837fc8668","side":"right"},{"sibling":"7465037b73da1c6d5a09594cc74ed4adc44a0c0cec0b8af8bf9d032be113e74f","side":"right"},{"sibling":"5e99ef044b2be7909dd96aee81514db76ba8f9c99fbda24af3b65cab39a1ad34","side":"right"},{"sibling":"bd78138f50c7c2cd7ea18c476126fa50e9d64c56fd172f68302e76b36ce87af6","side":"left"},{"sibling":"d343697805c8f4d7c4fdaed3423cc0b453fdf360c5ad7106d5f579c89d25631e","side":"right"},{"sibling":"6f1153ad830bc8202e6e1954981665e478138b53ce7dc91b62be904bade1d7fe","side":"right"},{"sibling":"8b0755e3ba26ea95e61e8d0e6362ce6cbe9991903d18e148b0cba214724147ec","side":"left"},{"sibling":"6c8b632bc887ad8683d2d08a66a66babfd1e8e43de4b1b760ed3f3d35d5b02e1","side":"left"},{"sibling":"879666fab72e779ab55d0564eaabd64b00534fd6bba7f18412c7f31f61ffd09f","side":"right"},{"sibling":"f40ccedd90c323817e961adc0a2e2db82b8aabe192b6c9d5a373ff988987b207","side":"right"},{"sibling":"b85ea61ae405eed84392a7b6b1eee5536f5a38d6b04070638e23ec7b71e3443a","side":"right"},{"sibling":"d415e6939aee710631f5062799379b547d2c3e3d9a68f263bbb5a693285ab2ca","side":"left"},{"sibling":"e5893793e3591ed7f5e58ca94ffcfba46bb30f69fb1c25d5ba8ef49eb99f9126","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"e3eecaf996dbe7229a7bb1d234c97aea97a252c5f7c89f8547b6d091db0f0e40","side":"right"},{"sibling":"55bcbd4da3e20d93931f7e58673f10232e81a5b1514d7396cb4b71e8f95788d0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":152106,"merkle_root":"ac5182c6f3dd09931f2a689df5f4be36df7b55e55bcf106e195671f5ed55fd8f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260526T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-26T05:37:34Z","sig_algorithm":"ed25519","signature":"a401243fbd2c077d29a623c6ef616c78fbd5c8ce1930afa7af7165b2386e5a8cf15d5094983a1c962e71b27b911a3f3ce09c8cfab9393be2ce5ce8ee6513da06","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1a70107754fc158289522e775ead02426bd35caa67e6fc33663520657a5f6580"}}