{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_f202f9d58183055b2e69eba715fe4a9ac65cd6596e65ee2716754ca822f387f0","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_f202f9d58183055b2e69eba715fe4a9ac65cd6596e65ee2716754ca822f387f0","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0a22d817fa9275599038b8c17506fd55cf221781023d6e551ca251f78bbb0b4d","published":"Mon, 29 Jun 2026 00:00:00 -0400","receipt_hash":"0a22d817fa9275599038b8c17506fd55cf221781023d6e551ca251f78bbb0b4d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0a22d817fa9275599038b8c17506fd55cf221781023d6e551ca251f78bbb0b4d","observed_at":"2026-06-29T04:44:03.414425Z","parent_run_hash":"36b5ab5c57ae76dc9e1a863501c4d38f172868cfae31b4ba5baf3caffaafb2c4","published":"Mon, 29 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.28166v1 Announce Type: new \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has significantly improved the reasoning capability of large language models, reaching expert or even superhuman performance in domains such as competition math. However, whether weaker agents and humans can actually harness this capability is far less certain, with RLVR documented to drift reasoning toward idiosyncratic patterns such as poor readability and language mixing. Tandem training is a recently introduced paradigm that targets this compatibility problem: a trained, stronger senior co-generates each rollout with a frozen, weaker junior, and the two are rewarded as a team, so the senior is pushed to reason in ways the junior can follow. Yet this paradigm has so far been demonstrated only in proof-of-concept settings, leaving open whether it scales to the long chains of thought of the modern RLVR pipeline. In this work, we propose Tandem Reinforcement Learning (TRL), which carr","title":"Tandem Reinforcement Learning with Verifiable Rewards","url":"https://arxiv.org/abs/2606.28166","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.28166v1 Announce Type: new \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has significantly improved the reasoning capability of large language models, reaching expert or even superhuman performance in domains such as competition math. However, whether weaker agents and humans can actually harness this capability is far less certain, with RLVR documented to drift reasoning toward idiosyncratic patterns such as poor readability and language mixing. Tandem training is a recently introduced paradigm that targets this compatibility problem: a trained, stronger senior co-generates each rollout with a frozen, weaker junior, and the two are rewarded as a team, so the senior is pushed to reason in ways the junior can follow. Yet this paradigm has so far been demonstrated only in proof-of-concept settings, leaving open whether it scales to the long chains of thought of the modern RLVR pipeline. In this work, we propose Tandem Reinforcement Learning (TRL), which carr","title":"Tandem Reinforcement Learning with Verifiable Rewards","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-29T04:44:03Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.28166"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:2620c3ce6d57e0802356d270dd4e8925eeb3cf8d7388855ff7876c590c78079c732c3382c80178901abe302ad7a65b730e3443632c44b14c7361978afe30c605","signer":"crovia.substrate","subject":{"observed_at":"2026-06-29T04:44:03Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.28166"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"5f8a565b525167fa72f3e8576a7656958bd0d891d6972c41ed265c557772b46b","leaf_index":261333,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"3da4ef3116518c3ad4de9dbb7cc735a97de5456d4b73f21ed6c8fbf0ce55663f","side":"left"},{"sibling":"44488d1934ace5782be6ca24007e991a72c8878fde42e5fc8883084556b1c10b","side":"right"},{"sibling":"51079b654c4bea850d960c46358aac2eb04c95e53ab7db264aa514e2996f5c42","side":"left"},{"sibling":"c6f7d006a179a750aaf5f08bf6e4506b77ebb1539e97da100dd963cc2c85c4b6","side":"right"},{"sibling":"d1e524ea5dbee522820fbf2c254ffc325584baf66a40331b2ca26df9f4152a38","side":"left"},{"sibling":"29e91c8cbbd7b5d986a89514acfb133bc4553933f559284c7b35fe66f2970db0","side":"right"},{"sibling":"ab8f67f857fa81c6ddd6431b5feff1b6dea6f58d2d6384543bad28001936ae23","side":"left"},{"sibling":"daf5279d003dd81baccfe21e3bfac0a3b468c08210a2f76237353b11695cdd73","side":"left"},{"sibling":"3698d5368a778dfd474a1084879c71ccc04d1602253019e48bab7329c680972f","side":"right"},{"sibling":"9f9daa9d12e65b219f34c92aec45450536b79a42b8892050d66961432ae28ed1","side":"right"},{"sibling":"b5725d7b0807dc6da32d9788f20057fa8726be38d30a9ebdabc605ae92739122","side":"left"},{"sibling":"e321b2cac14cbe28f76ccb7938249a40ff60cd5d2128b5634be046ea10e984b8","side":"left"},{"sibling":"5900dc6c7d13855af9d0385baf1691ec386df33e450c422af1cabe0a36e40ad8","side":"left"},{"sibling":"ae636ddee98c71ab7a7dc55ddfab70c7f710a2b6abfdf7a8b5d16a4017d1c0d1","side":"left"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":261662,"merkle_root":"aa8865c239aa2eb6c8aa7c6250f56b3cd5709854a8a07f6a29eb4ddd8802cb6f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260629T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-29T05:38:02Z","sig_algorithm":"ed25519","signature":"476329233e82fb35fba2552ddc5d1d75b2bdd8513bbd281e9c40a0b8e475df374a62dcd8b456b0c5e8815984f5b4bf0983ae95d2cf4d7412ebb13a433b933c0a","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_f202f9d58183055b2e69eba715fe4a9ac65cd6596e65ee2716754ca822f387f0"}}