{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_96e2e783c283eedfc23fbc32663b89ad8f853b7de7535ddb00444b8dd392f7dc","bitcoin_anchor":{"bitcoin_attestations":["bitcoin_block_949451"],"calendar_attestations":["https://finney.calendar.eternitywall.com","https://btc.calendar.catallaxy.com","https://alice.btc.calendar.opentimestamps.org","https://bob.btc.calendar.opentimestamps.org"],"ots_url":"/registry/data/substrate/anchors/77fc9c28fae777b81da5b495b3115474df6592dfac590333213d3bdf8b94a9b3.ots","stamped_at":"2026-05-15T03:00:03Z","status":"bitcoin"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_96e2e783c283eedfc23fbc32663b89ad8f853b7de7535ddb00444b8dd392f7dc","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"5e12cfa66b373658b19af71e035c7361a1e7b5487a8f2c842ca467d2fd6b3187","published":"Wed, 06 May 2026 00:00:00 -0400","receipt_hash":"5e12cfa66b373658b19af71e035c7361a1e7b5487a8f2c842ca467d2fd6b3187","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"5e12cfa66b373658b19af71e035c7361a1e7b5487a8f2c842ca467d2fd6b3187","observed_at":"2026-05-06T04:43:18.518819Z","parent_run_hash":"184209fc0f2ebdc4b721e1a74f74ab17637dab3344682b6e6bd1f3e5e8f2cd00","published":"Wed, 06 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.05134v2 Announce Type: replace-cross \nAbstract: We study how reasoning evolves in a language model -- from supervised fine-tuning (SFT) to reinforcement learning (RL) -- by analyzing how a set of theoretically-inspired datasets influences language model performance in chess. We find that fine-tuning a model to directly predict the best move leads to effective RL and the strongest downstream performance -- however, the RL stage elicits \\textit{unfaithful} reasoning (reasoning inconsistent with the chosen move). Alternatively, training on multi-move trajectories yields comparable downstream performance with faithful reasoning and more stable RL. We analyze multiple qualitative and quantitative measures and highlight how these evolve from SFT through RL; we find several SFT-checkpoint metrics -- spanning evaluation performance, hallucination rates, and reasoning quality -- to be predictive of post-RL model performance. Finally, we ground our results with an experiment measuring","title":"How Reasoning Evolves from Post-Training Data: An Empirical Study Using Chess","url":"https://arxiv.org/abs/2604.05134","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.05134v2 Announce Type: replace-cross \nAbstract: We study how reasoning evolves in a language model -- from supervised fine-tuning (SFT) to reinforcement learning (RL) -- by analyzing how a set of theoretically-inspired datasets influences language model performance in chess. We find that fine-tuning a model to directly predict the best move leads to effective RL and the strongest downstream performance -- however, the RL stage elicits \\textit{unfaithful} reasoning (reasoning inconsistent with the chosen move). Alternatively, training on multi-move trajectories yields comparable downstream performance with faithful reasoning and more stable RL. We analyze multiple qualitative and quantitative measures and highlight how these evolve from SFT through RL; we find several SFT-checkpoint metrics -- spanning evaluation performance, hallucination rates, and reasoning quality -- to be predictive of post-RL model performance. Finally, we ground our results with an experiment measuring","title":"How Reasoning Evolves from Post-Training Data: An Empirical Study Using Chess","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-06T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.05134"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:6d5f111d3841a4202aee6124605520f0cfe8c08c9f28f6e03573af7395916f5b8e4b47b98c7e1a067e83e438eff9411506da62ba3b6203dc0d161317a97f3507","signer":"crovia.substrate","subject":{"observed_at":"2026-05-06T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.05134"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"50e1993845489bb683fad5776b4470f1ebbd323abbe7c48811416366ef7f6fba","leaf_index":116576,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"ecd1858c91fb606d907c4529e782722feb7905f2c04576aeb2a306658c1c18a9","side":"right"},{"sibling":"d2d273b21968eeca25c78af029e332143f1d4620727ec7f4f3f9cc5ac04d031a","side":"right"},{"sibling":"f6fab91e40f5e69101cff1d997db394ef6052693e30b897118f91fbbd0f52099","side":"right"},{"sibling":"e6f8b1006344e313909e427e64e0fb1c8f4ab79348259448984e6ae3dc917b03","side":"right"},{"sibling":"ed3e93f63567e2ffec336fdfc782e01efee6c753431c1e8856b3c5e94104301a","side":"right"},{"sibling":"492506bf2649d464e65df022bc9e61badd6bd5efa29ecac6bf0d07ecaeb4d90d","side":"left"},{"sibling":"dd4277f33b532244672f719abb849e200a307f7c7efa590e2eb41dbbb1e7844f","side":"left"},{"sibling":"6ee45885d9cd9ed12596a458214030b8197200a928104b38bdab7a9d649f44f8","side":"right"},{"sibling":"5574ba66329e909ad4ff093e3491b05d5aa4f4ee91405ec5ecdb639bea7d87ab","side":"left"},{"sibling":"c750384c973eaf46aa237cdabe7a75729862dd4101dca10cb8e0626e0ec2d34a","side":"left"},{"sibling":"ddc060ac400459417799f688c75d5b271636afa40d89ec7fc98652473a8061f6","side":"left"},{"sibling":"282afa51266e47629e34d808a360bbb276348bcc5a24bdcc93e83a76e580293e","side":"right"},{"sibling":"05f89b32c00462e60adf95c1fe4579cdc2791b36e8b17573d8f3b5fd5da95a0b","side":"right"},{"sibling":"8ccd9937a2c0d5c04044d07d1557791b7d07bb31eac41a39a675608d44b38f23","side":"right"},{"sibling":"3a5e69cf0803f4c91f3895ed7c9a95748fef240bec4422e167c05300f79f06c0","side":"left"},{"sibling":"f2817ab288b5324fe49770372c7a10f33f7cd11005f8d4c0a730316f5229dc98","side":"left"},{"sibling":"725fac972e772ca0dc598810ea1abc70df472f72d2d6ab8a0baee2b80e5d2f4c","side":"left"},{"sibling":"98fc57dfef8873b512edc8340f7181df57302bb96777625e072235c62d7c5895","side":"right"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":134292,"merkle_root":"77fc9c28fae777b81da5b495b3115474df6592dfac590333213d3bdf8b94a9b3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260515T023701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-15T02:37:25Z","sig_algorithm":"ed25519","signature":"68107a834b00b24f5d4501e5ec727445311f132a486567ecc4c72a4e6dff24c8c21f2de3105293353ba5fdbe370d032819af6aa70f694e2e39b6af6737507009","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_96e2e783c283eedfc23fbc32663b89ad8f853b7de7535ddb00444b8dd392f7dc"}}