{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_add873b244c83c4f2102edb280de36c593b05dbf2614a9ea8e8e5220e1c89aef","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_add873b244c83c4f2102edb280de36c593b05dbf2614a9ea8e8e5220e1c89aef","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"fafc132e242ca2f0a2d98bb41df9c8dff85d8d4b965d6992ca350185d020de39","published":"Thu, 04 Jun 2026 00:00:00 -0400","receipt_hash":"fafc132e242ca2f0a2d98bb41df9c8dff85d8d4b965d6992ca350185d020de39","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"fafc132e242ca2f0a2d98bb41df9c8dff85d8d4b965d6992ca350185d020de39","observed_at":"2026-06-04T04:43:08.243501Z","parent_run_hash":"298818240313a3c9ce3ace3750dbf55845013dc3bc59aeced6331eccefdb61ac","published":"Thu, 04 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2601.15158v4 Announce Type: replace-cross \nAbstract: Transformers trained via Reinforcement Learning (RL) with outcome-based supervision can spontaneously develop the ability to generate intermediate reasoning steps (Chain-of-Thought). Yet the mechanism by which sparse rewards drive policy gradient to discover such systematic reasoning remains poorly understood. We address this by analyzing the policy gradient dynamics of single-layer Transformers on a synthetic graph traversal task that cannot be solved without Chain-of-Thought but admits a simple iterative solution. We prove that despite training solely on final-answer correctness, policy gradient drives the Transformer to converge to a structured, interpretable algorithm that iteratively traverses the graph vertex-by-vertex. We characterize the distributional properties required for this emergence, identifying the critical role of \"simple examples\": instances requiring fewer reasoning steps. When the training distribution plac","title":"Outcome-Based RL Provably Leads Transformers to Reason, but Only With the Right Data","url":"https://arxiv.org/abs/2601.15158","vendor":"arxiv_cs_ai"},"summary":"arXiv:2601.15158v4 Announce Type: replace-cross \nAbstract: Transformers trained via Reinforcement Learning (RL) with outcome-based supervision can spontaneously develop the ability to generate intermediate reasoning steps (Chain-of-Thought). Yet the mechanism by which sparse rewards drive policy gradient to discover such systematic reasoning remains poorly understood. We address this by analyzing the policy gradient dynamics of single-layer Transformers on a synthetic graph traversal task that cannot be solved without Chain-of-Thought but admits a simple iterative solution. We prove that despite training solely on final-answer correctness, policy gradient drives the Transformer to converge to a structured, interpretable algorithm that iteratively traverses the graph vertex-by-vertex. We characterize the distributional properties required for this emergence, identifying the critical role of \"simple examples\": instances requiring fewer reasoning steps. When the training distribution plac","title":"Outcome-Based RL Provably Leads Transformers to Reason, but Only With the Right Data","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-04T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2601.15158"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:e3eaa59913c908029eb37cca4d19b2be5038995d5e5b8ee020ddc4e12348dfee0e4a2073087ab157c0a3e23afa15142ebbe1c19e9bdaf47027ea1e89c6bc400e","signer":"crovia.substrate","subject":{"observed_at":"2026-06-04T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2601.15158"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c64d17da52ae628509cddf2feb57ce7f6e2319013bd3f1f03d9b6b6b9069351a","leaf_index":212831,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5c7db8e999c41d0319440fb590b4efca2af1152a3cedff0d01eda18f26766491","side":"left"},{"sibling":"4308b72de073bf065abebe7ce73e5f64a40807dc909b4bfb4fc5293816b5e537","side":"left"},{"sibling":"2481a5dafc3eb171d6bbf28443bc884b3c1c9fe1c7c5855854baa99d91e6d165","side":"left"},{"sibling":"9a5fb91020503dc395f205bdef1f898d299612977fc53df085942be231d0a328","side":"left"},{"sibling":"45410d05b96813c142cc13502a7952136d885c32876ba0f81d1e394fe2d156dc","side":"left"},{"sibling":"b043ef4c24a378c518cd937d028d5345784f2a0844965c6fb86a6480a4862c3d","side":"right"},{"sibling":"a4d720a21c128acd66da32542f5a6516eadd1108045e73ba02163baa0cf221fc","side":"left"},{"sibling":"5d10c089cc50333cc6f5ec8f5c37b251f9cdd4252a9604bb46ec83ac720abdd9","side":"right"},{"sibling":"b4e8a5af3a9fdc4bb815270cd541ee4baf09fb1d9e2cccd75c7346558e1f1e4c","side":"left"},{"sibling":"cfb460164a914d1f96a36aa17124b45bb5417829d2adbe5ca48375dbd842ec44","side":"left"},{"sibling":"a84ebc8e894a9893ee34d1afc942d15f28243b926f1e5f9d7f10a3de1e795262","side":"left"},{"sibling":"f8f6bd9da448fa097e2115b71146f61691d9f8807aca291c87c633192a7224e9","side":"left"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"422bcf7e281ca3a607f3726a5e6b8fabdb85e8b32199b4a86356998e260a0b34","side":"right"},{"sibling":"54a99163a4a62374c3ca6fb46294222f1d4b1a9d0b636e27256b0e093e98239a","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":213053,"merkle_root":"19d104b92c4d7299881c447fb8422611fc9cb8615d37a538959e34a2da7ef55f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260604T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-04T05:37:49Z","sig_algorithm":"ed25519","signature":"630748e88645187aa3b4d4cb8c872cc146180d0301e4c6656f15f99081d7776432715cea2a997b32b7b3a5216f215d19c1b536c0b093100ec843fa0bd65f2101","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_add873b244c83c4f2102edb280de36c593b05dbf2614a9ea8e8e5220e1c89aef"}}