{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_60d9a46aba50eacc6f8f5ad98cbc69a8d90336511e0de75034edf6b326969369","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_60d9a46aba50eacc6f8f5ad98cbc69a8d90336511e0de75034edf6b326969369","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"afd4257c1e8b51852d6854cb0eb17e08e8db1ca783a0629cae0072d4fdbd8b63","published":"Thu, 28 May 2026 00:00:00 -0400","receipt_hash":"afd4257c1e8b51852d6854cb0eb17e08e8db1ca783a0629cae0072d4fdbd8b63","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"afd4257c1e8b51852d6854cb0eb17e08e8db1ca783a0629cae0072d4fdbd8b63","observed_at":"2026-05-28T04:43:38.862500Z","parent_run_hash":"58f8b4a134069e0a15ea3949252489597eb86dd27c9ca3fb15c6fb838ce49ef3","published":"Thu, 28 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.27765v1 Announce Type: cross \nAbstract: Self-Distillation Policy Optimization (SDPO) provides dense token-level credit assignment for reinforcement learning with large language models by leveraging the model's own feedback-conditioned predictions as a self-teacher. Unlike GRPO, however, whose group-relative advantage naturally concentrates learning on a sweet spot of intermediate-difficulty questions, SDPO's KL-based advantage lacks an implicit notion of difficulty awareness.\n  We analyze this gap through the lens of GRPO's advantage normalization. Extending the learnability framework to normalized rewards, we show that normalization absorbs the variance term $p(1-p)$, equalizing leading-order learnability across questions and leaving $\\sqrt{p(1-p)}$ as the sole residual scaling factor in the per-question gradient. This analysis yields a simple prescription: weight each question's SDPO loss by $[\\hat{p}(1-\\hat{p})]^{1/2}$, resulting in SC-SDPO, a scale-consistent variant of ","title":"Restoring the Sweet Spot: Pass-Rate Weighted Self-Distillation for LLM Reasoning","url":"https://arxiv.org/abs/2605.27765","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.27765v1 Announce Type: cross \nAbstract: Self-Distillation Policy Optimization (SDPO) provides dense token-level credit assignment for reinforcement learning with large language models by leveraging the model's own feedback-conditioned predictions as a self-teacher. Unlike GRPO, however, whose group-relative advantage naturally concentrates learning on a sweet spot of intermediate-difficulty questions, SDPO's KL-based advantage lacks an implicit notion of difficulty awareness.\n  We analyze this gap through the lens of GRPO's advantage normalization. Extending the learnability framework to normalized rewards, we show that normalization absorbs the variance term $p(1-p)$, equalizing leading-order learnability across questions and leaving $\\sqrt{p(1-p)}$ as the sole residual scaling factor in the per-question gradient. This analysis yields a simple prescription: weight each question's SDPO loss by $[\\hat{p}(1-\\hat{p})]^{1/2}$, resulting in SC-SDPO, a scale-consistent variant of ","title":"Restoring the Sweet Spot: Pass-Rate Weighted Self-Distillation for LLM Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-28T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.27765"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:effba043f3990a13c41d99d4ebbba41a6486543f602ef4654bfaf96dbc970e2d69c315cf37680984c32eb465ba34b3a4ee623bd6a7d34ac9dd1a72ca144ed309","signer":"crovia.substrate","subject":{"observed_at":"2026-05-28T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.27765"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"2274faa3f8838639e509c0c8e5af721993ddd3f2c7523d406e4d638f380a911b","leaf_index":155816,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fec04c775daa406c9f42a3a59e8619bd28c94c54ddf63db3e574ae20c9b6ed29","side":"right"},{"sibling":"2d7ae7555fa2e4cfca6cb5ae47fb5df3f5963031585102e7db43bb5b22e72322","side":"right"},{"sibling":"9515312cd8ba67ea449b6e12e8e05ed7e0c6531d286cc7ff1939fcf01f2794dd","side":"right"},{"sibling":"98b2315816614113be63bae11cc784b8f8ad4bf6ea6193b0af56341681be94f8","side":"left"},{"sibling":"e33b06cbdc5cee06c8d05bee8df4e75bb5189812bafae3a414bb0e31d7fd011a","side":"right"},{"sibling":"7e53d22770bf5e38d6d023fbbe1e1820571bc78d004bac5ed23614ccf9c29566","side":"left"},{"sibling":"ee7da65b5a86af06f9a3335d17a9d3cfc91a80b40810cef933dac1f3ef2460ba","side":"right"},{"sibling":"753018fa645affe1bfe31d1c496292a6e72167116d741961e921b85a693e4153","side":"left"},{"sibling":"0adb35dd3ad9bfc4a781c02606acf0a3ce1e28e1a88066209cc1a35814e790ad","side":"right"},{"sibling":"f292d3278e493ec60902181b8c0bd5c89c0a1168ef928222fe6161287982f7f0","side":"right"},{"sibling":"2209295faf1a5bf51c97c6fd5a839a8181a4ea44f490420381530f35df7d9b2f","side":"right"},{"sibling":"5784576a15214ea9fc3569e6e1cff1ef443c0b1fc0d036028489088af089de27","side":"right"},{"sibling":"311772ec218efcb2da5a337f9e9f042fe1cc0028643adb0a354787e4ea7911b7","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"7b6f0bea4291a5e63574dfca9aa0f9756f450c9478c3d07474d39a7ababb51f9","side":"right"},{"sibling":"1d39fe14b21e2ebbfb87e882423b24ee9469eae1e4c77af5b799ac4db9537467","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":156177,"merkle_root":"c5705a0243d16afd8b1ebfd731b7aa304079c442c2a7906493c5bbed374c69ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260528T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-28T05:37:36Z","sig_algorithm":"ed25519","signature":"f087e13febc8bb6a2e0812610de64cebc65be92915518d9c4b230799c3b161839b04c4eb1b02741f938f35545a76ab76555b04c782bdc2f9a44852d171d65909","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_60d9a46aba50eacc6f8f5ad98cbc69a8d90336511e0de75034edf6b326969369"}}