{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d1bc80a656281483cedfec6816f9942893e15fc9ca534750fecb0aa2f09b99f7","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d1bc80a656281483cedfec6816f9942893e15fc9ca534750fecb0aa2f09b99f7","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a5f930e3ae5dc3288a86d61af103db1ef28b2bd072c3b245dbfc24b94de65254","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"a5f930e3ae5dc3288a86d61af103db1ef28b2bd072c3b245dbfc24b94de65254","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a5f930e3ae5dc3288a86d61af103db1ef28b2bd072c3b245dbfc24b94de65254","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.01281v1 Announce Type: cross \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has emerged as a powerful paradigm for enhancing the reasoning capabilities of large language models (LLMs). However, its effectiveness is substantially hindered by the prevalence of ineffective training data: many sampled prompts yield response groups that are either entirely correct or entirely incorrect, resulting in zero-variance rewards and limited learning signals. Recent state-of-the-art methods address this issue through extensive LLM rollouts to filter ineffective samples, but at the cost of considerable computational overhead. Alternative approaches, including predictive sampling and trajectory replay, aim to improve data efficiency but often remain insufficient and may introduce additional issues such as systematic bias or suboptimal constraints. To address these limitations, we propose Group Prioritized Off-Policy Optimization (POPO), a simple yet effective framework tha","title":"RLVR without Ineffective Samples: Group Prioritized Off-Policy Optimization for LLM Reasoning","url":"https://arxiv.org/abs/2606.01281","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.01281v1 Announce Type: cross \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has emerged as a powerful paradigm for enhancing the reasoning capabilities of large language models (LLMs). However, its effectiveness is substantially hindered by the prevalence of ineffective training data: many sampled prompts yield response groups that are either entirely correct or entirely incorrect, resulting in zero-variance rewards and limited learning signals. Recent state-of-the-art methods address this issue through extensive LLM rollouts to filter ineffective samples, but at the cost of considerable computational overhead. Alternative approaches, including predictive sampling and trajectory replay, aim to improve data efficiency but often remain insufficient and may introduce additional issues such as systematic bias or suboptimal constraints. To address these limitations, we propose Group Prioritized Off-Policy Optimization (POPO), a simple yet effective framework tha","title":"RLVR without Ineffective Samples: Group Prioritized Off-Policy Optimization for LLM Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.01281"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:37e3773b7178e35b586b9c761b35d4d1ed0bc0bbee7008246ffa0c1b795f58f0ebca12ea4e2ac6fd7d7f24c0a6c328972af30b57d08896a7ee02fdb6c01d8c05","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.01281"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8a8075d7b26f48e81f2d003e9ff6b107ab2e8faacd68f2272d159eab6684410c","leaf_index":205585,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5f525ed439668d6d0fcf5125a895b6241ea78078813b43dd8418f25eaf82dbd6","side":"left"},{"sibling":"ac0e473dea46694e452117c2a88867756cf8bfef0c1b7c82c758a20790397dcd","side":"right"},{"sibling":"d40fcdbc0a77603bab0f9b189bd61082d44a7c93f37cc40d9c07c00429f9bf52","side":"right"},{"sibling":"0a566755aef92f4b53207d9eb90e0d5d376774219bb828c74bf6ff37d295fa53","side":"right"},{"sibling":"2e8f87a4c62fd8a094157cc68e564f8612b8e5a0bda22b358b018288b7a6780a","side":"left"},{"sibling":"bf8fe4fbdd715e3289029a0a2b189d8a48f5c617f2f67bb5d06b20d4944520a0","side":"right"},{"sibling":"1e82068bd784779754cf4c7ebad995756b8568c4c3050604d6f3f530fe4825b9","side":"right"},{"sibling":"41924649fb2045a8eb1c76bcccd1d2cd8c7e0ab0bba7b07a79bf25a0c1677263","side":"right"},{"sibling":"21332ee1c3955bf1803955bf945ab0966ce5e0aa48d8642320cdbc044a72b935","side":"left"},{"sibling":"6620a5008acf732cd3e57b0f2d1437293e88366c2e5b175ebeb7d796fc1e0c62","side":"left"},{"sibling":"a82575bfb494af7afcc13aaae718afa6f74030f09d71b02819ea25efd4186fc4","side":"right"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d1bc80a656281483cedfec6816f9942893e15fc9ca534750fecb0aa2f09b99f7"}}