{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_89deba108539bca683a7031ef8754165a0580c6f57af17d143feb89f4d15f855","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_89deba108539bca683a7031ef8754165a0580c6f57af17d143feb89f4d15f855","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"aa36b1030c2dffed63b50ff58edcbf0bb37beaea1a8d3b71dcc3276bb5e51faf","published":"Fri, 29 May 2026 00:00:00 -0400","receipt_hash":"aa36b1030c2dffed63b50ff58edcbf0bb37beaea1a8d3b71dcc3276bb5e51faf","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"aa36b1030c2dffed63b50ff58edcbf0bb37beaea1a8d3b71dcc3276bb5e51faf","observed_at":"2026-05-29T04:43:58.478092Z","parent_run_hash":"0fcd87efcfe67ccb9952f747541debc16793919a4d20fd71ca0ad5516a0a13ee","published":"Fri, 29 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.13230v2 Announce Type: replace-cross \nAbstract: On-policy distillation (OPD) has become a promising paradigm for reasoning-oriented post-training of large language models (LLMs), especially when combined with reinforcement learning from verifiable rewards (RLVR). Existing OPD methods rely on reverse KL (RKL)-based teacher supervision over trajectories sampled from the student policy. However, we identify a critical limitation: under large teacher--student policy divergence, RL-driven exploration often produces trajectories outside the teacher distribution, resulting in uninformative negative feedback. To address this, we propose Teacher-Guided Policy Optimization (TGPO), an on-policy reasoning distillation method that remains effective under large policy divergence settings. Rather than relying solely on evaluative supervision, TGPO uses teacher to directly guide token level generation conditioning on student-generated contexts; together with RLVR-style trajectory level rewa","title":"Teacher-Guided Policy Optimization for On-Policy Reasoning Distillation under Large Policy Divergence","url":"https://arxiv.org/abs/2605.13230","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.13230v2 Announce Type: replace-cross \nAbstract: On-policy distillation (OPD) has become a promising paradigm for reasoning-oriented post-training of large language models (LLMs), especially when combined with reinforcement learning from verifiable rewards (RLVR). Existing OPD methods rely on reverse KL (RKL)-based teacher supervision over trajectories sampled from the student policy. However, we identify a critical limitation: under large teacher--student policy divergence, RL-driven exploration often produces trajectories outside the teacher distribution, resulting in uninformative negative feedback. To address this, we propose Teacher-Guided Policy Optimization (TGPO), an on-policy reasoning distillation method that remains effective under large policy divergence settings. Rather than relying solely on evaluative supervision, TGPO uses teacher to directly guide token level generation conditioning on student-generated contexts; together with RLVR-style trajectory level rewa","title":"Teacher-Guided Policy Optimization for On-Policy Reasoning Distillation under Large Policy Divergence","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-29T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.13230"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1a931d962522572e4539c8ba9bb02ea38b43fd58ac83e12ab97274ed49032c19489e534a0208d62155f0e1de382bbccdc508c602575e81c07776ccfe9d1db60c","signer":"crovia.substrate","subject":{"observed_at":"2026-05-29T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.13230"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"02ddc5ae5441d02895bb256d0f9ee7c32a93f6ff7b0c8ae2d3df389f7f0349c0","leaf_index":158167,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"64778eaa5224f0a073f00795cb9ce3b8c60c549d338e928c1853dbef0994dbdd","side":"left"},{"sibling":"263f63617e1975ac5f84e80cd9aa32fb323f809a5c70154a08a5e37c11be02b9","side":"left"},{"sibling":"e1da45ac65d03c8a3cee763f690b1cea0ef1d1c7aa542f1f6d0496cf41ad9d14","side":"left"},{"sibling":"48d3f688053cbb21527f1e8fddd762c8df6247ad6217c2c50c9c7a27102e6a85","side":"right"},{"sibling":"e26abcb1c5e9dd1cef706a8743245bc5fb03958c119f061fe9f2e44c6a090920","side":"left"},{"sibling":"6cdda38b15191f258aeb0dcad31a49823191d9be28cb42bfa3793dfd48952b6c","side":"right"},{"sibling":"2496d8181c3184c4ea7973d99293dedbb754d677261137e3c8c5546635804645","side":"left"},{"sibling":"463d4726f68bda2bbf724c72553fd53d88ea99a3d4a713f9b8a1b2d9fde12933","side":"left"},{"sibling":"002471ff5893cd2a2890119cbfe4a9456bfb46cf31fa6aee377650afe6ae0c92","side":"left"},{"sibling":"1dcc44e23fbb0218b13591e4b584eca3600dcf365769cb741e0ecd33b25b8c56","side":"right"},{"sibling":"fcf16a6f44025801f5b83e928acce764352ddbc06ae3043d8cde9a409933e6d8","side":"right"},{"sibling":"78982294dee68f9db9288c64d7e507c7865fda96e1b6f7ccff5c8bb152e93c49","side":"left"},{"sibling":"995b421824624a8282c7f44e64c64ee35344800f477ae1845b41be14d3fab94c","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"eef0e8906a749d3470f89beeedc723f37a5737010bbb0dcc7cf91515338e5a3e","side":"right"},{"sibling":"1a07e481a9407d71aad078ce854cdeee362163c887fe10f889b0ecf0b5e749ad","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":158251,"merkle_root":"485e6b31fe60c8beba5b394808c7e4c32448b2ff65c2482c480ca0e2a2eda718","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260529T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-29T05:37:37Z","sig_algorithm":"ed25519","signature":"bbf9f005201182fce4f9d94c7a9d01a508b56daf7d9611bd73514f5f616bc059d0e5e1f2edfc95716e6fe08ef5fae38b558cbf7f2fd8f9d5dfe4c34a54c83005","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_89deba108539bca683a7031ef8754165a0580c6f57af17d143feb89f4d15f855"}}