{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_805d329d820e801fd21dcbd52162112da5783b8d119ece69232e6de3866b5dc1","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_805d329d820e801fd21dcbd52162112da5783b8d119ece69232e6de3866b5dc1","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"2e3fa9b11d635925e09f155266cde69f8489b63782311a456a83d8b53a5d5826","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"2e3fa9b11d635925e09f155266cde69f8489b63782311a456a83d8b53a5d5826","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"2e3fa9b11d635925e09f155266cde69f8489b63782311a456a83d8b53a5d5826","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.01462v1 Announce Type: new \nAbstract: Studies of human reasoning have shown that people are typically stronger at evaluating reasoning than producing it from scratch. In contrast, large reasoning models (LRMs) are trained to excel at producing long chains of reasoning to solve complex problems. How then do LRMs perform at evaluating reasons? We investigate this with the Valid-Answer-Invalid-Reasoning (VAIR) dataset: math problems and solutions with trivial reasoning flaws but valid answers, designed to isolate reasoning evaluation from the confound of reasoning production. Unlike humans, who we find are only 6% worse at grading than solving such problems, we find a substantial production-evaluation gap in LRMs: frontier models score as low as 48% when evaluating VAIR solutions, despite near-perfect solution production.\n  Why this enigma? Through chain-of-thought (CoT) analysis, we find evidence of an answer confirmation bias: LRMs often produce then check for the correct ans","title":"An Enigma of Artificial Reason: Investigating the Production-Evaluation Gap in Large Reasoning Models","url":"https://arxiv.org/abs/2606.01462","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.01462v1 Announce Type: new \nAbstract: Studies of human reasoning have shown that people are typically stronger at evaluating reasoning than producing it from scratch. In contrast, large reasoning models (LRMs) are trained to excel at producing long chains of reasoning to solve complex problems. How then do LRMs perform at evaluating reasons? We investigate this with the Valid-Answer-Invalid-Reasoning (VAIR) dataset: math problems and solutions with trivial reasoning flaws but valid answers, designed to isolate reasoning evaluation from the confound of reasoning production. Unlike humans, who we find are only 6% worse at grading than solving such problems, we find a substantial production-evaluation gap in LRMs: frontier models score as low as 48% when evaluating VAIR solutions, despite near-perfect solution production.\n  Why this enigma? Through chain-of-thought (CoT) analysis, we find evidence of an answer confirmation bias: LRMs often produce then check for the correct ans","title":"An Enigma of Artificial Reason: Investigating the Production-Evaluation Gap in Large Reasoning Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.01462"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:196237e98dd6a3cbe9334c03972143045760a9f51046815e7b53d7f5539a22ece75b94dd1225c5f1a827d9ffb902a391d452b0032ba9e7753b35a8ca7d2fac0a","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.01462"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"83b1b4e80cc0cf52dc8f00cf523a464ca3869601a67723df935545c0c8310e20","leaf_index":205263,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5909ca2d16cf9670f49d3bafafbe6a6ca768e88ee34018dc4c737591ebb0ed3a","side":"left"},{"sibling":"71aca3d9551f2645a1e04e0cc87166af1e632fcde34b1e06d3367337dbb40ccb","side":"left"},{"sibling":"af357721514b4d6fb3dd22abd6cb04b134f9d1d02c8dedbc921d95f98f33edd4","side":"left"},{"sibling":"af913cb21b99afece2a118c0b31e970ec8ac0713abc4dfb3df5d598d8bf29584","side":"left"},{"sibling":"6e1a099d978c3bb57e50c9e903f232de57c45785ac96f20522859199811b295a","side":"right"},{"sibling":"a3ebb4a957dd61f983d05bf5292cb0950e18a3c7b751f41965168e814261e275","side":"right"},{"sibling":"09faabc6eb084c1a3194a3270e209c2e05c1c91adfaf601ce9892ebd8096ea74","side":"left"},{"sibling":"d610901cb65aa4d03839583811946eca8c0da378d2eaae1c73d15f7314c9374b","side":"left"},{"sibling":"3fed4154aebacb68ca58546a945dace8a8d7d2176af23d8fb700f153f2df448f","side":"left"},{"sibling":"30520dbc0b3dbdd0b4502a5ecc782d8aa02c8659f12dfb9b79fd55de5c05a0c5","side":"right"},{"sibling":"a82575bfb494af7afcc13aaae718afa6f74030f09d71b02819ea25efd4186fc4","side":"right"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_805d329d820e801fd21dcbd52162112da5783b8d119ece69232e6de3866b5dc1"}}