{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_ae7a458fd0b73f222235648b457b6dadea403e707e8b39e93d85d5528ded886f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_ae7a458fd0b73f222235648b457b6dadea403e707e8b39e93d85d5528ded886f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"9ed6a65d0523c191a008bc67f490a27873818ef81561d7cc52a668217287ea1c","published":"Tue, 21 Jul 2026 00:00:00 -0400","receipt_hash":"9ed6a65d0523c191a008bc67f490a27873818ef81561d7cc52a668217287ea1c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"9ed6a65d0523c191a008bc67f490a27873818ef81561d7cc52a668217287ea1c","observed_at":"2026-07-21T04:43:35.036805Z","parent_run_hash":"03e944014de2697434479833d15ea9303e014945afc230ecc7f207824493b589","published":"Tue, 21 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.16205v1 Announce Type: new \nAbstract: Reinforcement learning with verifiable rewards has emerged as a standard approach for enhancing reasoning in large language models, which typically optimizes the policy by contrasting multiple self generated rollouts. However, we identify a critical support limited bottleneck in this paradigm: on challenging reasoning tasks, the target model's samples often exhibit semantic redundancy, converging into the same erroneous \"reasoning basins\" that offer negligible reward contrast for policy updates. In this paper, we propose to overcome this limitation through a weak to strong learning paradigm, where a policy's exploration is informed by a weaker but computationally efficient auxiliary model. We introduce W2SPO, an off policy RL method that injects short auxiliary segments often as brief as 8 tokens into intermediate target model trajectories and the target model then completes the reasoning path from these diverted states. Policy updates a","title":"It Takes 8 Tokens: Weak-to-Strong Off-Policy RL via Auxiliary Branches","url":"https://arxiv.org/abs/2607.16205","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.16205v1 Announce Type: new \nAbstract: Reinforcement learning with verifiable rewards has emerged as a standard approach for enhancing reasoning in large language models, which typically optimizes the policy by contrasting multiple self generated rollouts. However, we identify a critical support limited bottleneck in this paradigm: on challenging reasoning tasks, the target model's samples often exhibit semantic redundancy, converging into the same erroneous \"reasoning basins\" that offer negligible reward contrast for policy updates. In this paper, we propose to overcome this limitation through a weak to strong learning paradigm, where a policy's exploration is informed by a weaker but computationally efficient auxiliary model. We introduce W2SPO, an off policy RL method that injects short auxiliary segments often as brief as 8 tokens into intermediate target model trajectories and the target model then completes the reasoning path from these diverted states. Policy updates a","title":"It Takes 8 Tokens: Weak-to-Strong Off-Policy RL via Auxiliary Branches","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-21T04:43:35Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.16205"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:8d4b750fb57bfe2b94604fd16ab57245d8bf0048150242727475142717cdc7422d23f62d0eba67d58e854cc83e017ca229de204b41d1cb0bd26ec3b56f37d10d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-21T04:43:35Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.16205"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"9769aa0266bbcd8374f0e4448085976089d13930a33077ad0fed77eb611b0f8b","leaf_index":336520,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"0698d9df61979bc13f173943ff92bc1962cc1e703745d7e1f203c90cf7311319","side":"right"},{"sibling":"fc5bad4d04f8283a5c3c4b99dd7718617b954701b937c847ed84f66d53344ee7","side":"right"},{"sibling":"ee126835b71a1a5e8cb3ee44e1f4864e0e394391b182f6766cda788ed421542b","side":"right"},{"sibling":"d9023d8647d1692ac7434f694021b2dbc27525d9622687e8d28f1e7043ef1fdd","side":"left"},{"sibling":"55d8e23342c2870e9173db93b6167c1b6b61760de29d182d80cb27c924b24cd2","side":"right"},{"sibling":"a24ad8a95338872e5d75e9c9b830f6e3d66ef973c0f35acc3df53f6d02fb947b","side":"right"},{"sibling":"a2ec0c96e7ee9134cdcbfc3f3d7279dabbae2ae36687e5158bdb5f7e8fc549cb","side":"right"},{"sibling":"e6f9d6d6c760446a7b30dd4e30a28f58c0817f4bee09529ee009c470d86f5564","side":"left"},{"sibling":"38e5827f7c9f72ad34a2b97042f2fb5f7e868db899d074b7819af4f2209b9caa","side":"right"},{"sibling":"7b927551b5db06b6571913b4e6792ffcce5291a3eca5a0df4a3b6e296271105f","side":"left"},{"sibling":"b77a0b5ae4607c8fe6ba73449d46b35076e3dedc0c82a2c65a05780d42a7bc2e","side":"right"},{"sibling":"9eb5077edfb3dc553857d4794b925bfce117e0f8a1d049af5d0dd9026b470eef","side":"right"},{"sibling":"414b1a70fd1dcb25489a194714b97492b066684b15d0b7a48a176c4b9b5bc713","side":"right"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"613f015699131eb89bd755dee67133be95af25cf5f16c1c8ce4b99d963b8dd86","side":"right"},{"sibling":"a729b574b1135956436ded5eef1fe8f08014ff6a0729749d307ab1bca93fcdc9","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"4bf21052e085e8ac81f1dec1d2b310bd12bf948992de6177d12e9d2fda8d39f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":337144,"merkle_root":"5e969cc01afa67e4dbe5d37b712cdb10f4aa1fd74404e02eab724cf487c8d6d9","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260721T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-21T05:38:38Z","sig_algorithm":"ed25519","signature":"5c0c1a8dd2793d787ccd5e49e8b4d70eed136352555518589f05c74be357fc171702e42a76c3a556d90e3d51ff36cb3d292aaac83c66566de7b943f318bda50c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_ae7a458fd0b73f222235648b457b6dadea403e707e8b39e93d85d5528ded886f"}}