{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_5af14f94685a608070e2f3f5635359985acf10a302404912fa142a003497351e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_5af14f94685a608070e2f3f5635359985acf10a302404912fa142a003497351e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e5f480385c82bce1da6d326986eab8d666042190390d5c0beb8b9775c125ded4","published":"Mon, 01 Jun 2026 00:00:00 -0400","receipt_hash":"e5f480385c82bce1da6d326986eab8d666042190390d5c0beb8b9775c125ded4","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e5f480385c82bce1da6d326986eab8d666042190390d5c0beb8b9775c125ded4","observed_at":"2026-06-01T04:43:13.859018Z","parent_run_hash":"8993bbc535dae8c9669e099af3624cb39166b8d9bbfd66f26ae5c338cbb21be2","published":"Mon, 01 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.31228v1 Announce Type: cross \nAbstract: Reinforcement Learning with Verifiable Rewards is an effective route for post-training to strengthen the reasoning capability of large language models. However, as training proceeds, the learning signal can collapse thus makes the training gain become marginal and ineffective. Specifically, a growing fraction of prompts' rollouts become advantage-degenerated: all the self-generated rollouts show verified-success, making the standard deviation over their rewards be zero; accordingly each rollout's advantage becomes degenerated (zero) as well. Given such rollouts' advantages, the policy-gradient for model optimization eventually vanishes, capping the training performance. We argue that some of these rollouts still contain valuable learning signals but unfortunately omitted with the existing RLVR methods. In this paper, inspired through analyzing the entropy pattern behind golden trajectories produced by external expert models, we propose","title":"EchoRL: Reinforcement Learning via Rollout Echoing","url":"https://arxiv.org/abs/2605.31228","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.31228v1 Announce Type: cross \nAbstract: Reinforcement Learning with Verifiable Rewards is an effective route for post-training to strengthen the reasoning capability of large language models. However, as training proceeds, the learning signal can collapse thus makes the training gain become marginal and ineffective. Specifically, a growing fraction of prompts' rollouts become advantage-degenerated: all the self-generated rollouts show verified-success, making the standard deviation over their rewards be zero; accordingly each rollout's advantage becomes degenerated (zero) as well. Given such rollouts' advantages, the policy-gradient for model optimization eventually vanishes, capping the training performance. We argue that some of these rollouts still contain valuable learning signals but unfortunately omitted with the existing RLVR methods. In this paper, inspired through analyzing the entropy pattern behind golden trajectories produced by external expert models, we propose","title":"EchoRL: Reinforcement Learning via Rollout Echoing","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-01T04:43:13Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.31228"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4ca092bef778dfa7c6a9f1c4df7de0df12d5eada67b7f4e049853b4d2f3dad8defd8b93c81cd5d0cca8c184bebc1eb931ca8d0af29c3743854a8b99b30bdfa03","signer":"crovia.substrate","subject":{"observed_at":"2026-06-01T04:43:13Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.31228"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8d6da7bdd5dd8c8cdc82d1f163b0d22304b558fd4133933e6c28738839db4127","leaf_index":163943,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"efda4fe87d6d61e5d45d9bfbecb7c5311faf717d1df71f24b2e53226c44af228","side":"left"},{"sibling":"0f946407877fc494fcc45c4b0c0f5566919b4ba3adb1bea90e1626a58c7eadb1","side":"left"},{"sibling":"2ff948bde0087f3ac8ea48f17703489fa6118b6282e80e0017b5a04810767392","side":"left"},{"sibling":"7f05c9235d1dea34ae3efbb9808c06d282e0c4ac00ceca2fb2cc86b0cf1acd82","side":"right"},{"sibling":"cef08938d74b036bc06dc91c7f89a56ce945f6c5d06d51cd94af2775dc4aedf7","side":"right"},{"sibling":"5dba26cc7bd1015790fa05d7e607a4c27b9d0de614892f2cc19bab7f28b9256b","side":"left"},{"sibling":"a1ff84cd2c69f189b478824886ec6dfb932db86d85674c90e121b3e31e941af7","side":"left"},{"sibling":"c80f3e378be31136ba3005ae1d4e6fb73ee42b46d44ded3d5ec7b012545b00f5","side":"right"},{"sibling":"86fede29e507d01bfe35484a1579149e8e99b3527ff48559423f6056d11af46a","side":"right"},{"sibling":"fc601f0745c37fc5f7c249300e654db05d61bb9059885eeb0b0a5047b5d28408","side":"right"},{"sibling":"6ed9290cdae063f13bfeb71c4c5440595cabb61fd4b39225b9cc913ffb336dd7","side":"right"},{"sibling":"a0446b923d1ce90021e78edff07f6bfc7cc2a1326a565c5f0786b82a24dd0a2a","side":"right"},{"sibling":"e598fd53912c30e58ca8e58d7d8a338fe0f2ecb63fdd99225bc703c499c948ec","side":"right"},{"sibling":"fc4873333221ec8167697f75b6f6a8a08491a8cf18952defb65fb6d4958fa5e7","side":"right"},{"sibling":"05c8a827da2a05549ee3250310777009120c687885816bf6c7c74801bfaa346d","side":"right"},{"sibling":"5ea2f2dc9f046df723b6bd9932d61a9a3d80a76e79ce1b939b3d93ff79b5a91c","side":"left"},{"sibling":"ce41d9b82f34b16efd653dfb3552acc4e2512939e47903e5fc979fbed00c5764","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":164217,"merkle_root":"a1098816aea1b60b8fe37b62410469bc5024a2c335bbec4f6ef2add7875dbdf2","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260601T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-01T05:37:41Z","sig_algorithm":"ed25519","signature":"d7f91db1d54b9495c499440c2828f4bd53360555391ce6e25adea5183bc1fa0f697d80708a099d0b0429e6f8cb6c71e7fccf82acb3c84481149974fb26074708","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_5af14f94685a608070e2f3f5635359985acf10a302404912fa142a003497351e"}}