{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_06817900a0b85758f360d7175d22bee44dfa32ffbd81d7d006b66a67cd657a45","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_06817900a0b85758f360d7175d22bee44dfa32ffbd81d7d006b66a67cd657a45","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"c99c4a28a81fdadd6abdc843da0a76354239d50003586d70ceccb14d744fa10c","published":"Thu, 11 Jun 2026 00:00:00 -0400","receipt_hash":"c99c4a28a81fdadd6abdc843da0a76354239d50003586d70ceccb14d744fa10c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"c99c4a28a81fdadd6abdc843da0a76354239d50003586d70ceccb14d744fa10c","observed_at":"2026-06-11T04:43:37.662146Z","parent_run_hash":"5267801b61ae0d882196b5f37208a9a1633905a64ca7d933f1fa5075cd861491","published":"Thu, 11 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.11211v1 Announce Type: cross \nAbstract: The ability of large language models (LLMs) to express calibrated uncertainty is important for safe deployment. Chain-of-thought (CoT) reasoning is widely used to improve accuracy and reliability, but its effect on calibration is not fully understood. We show that this picture is incomplete: in some settings, increasing the reasoning budget beyond a task-specific threshold can cause models to become systematically overconfident, assigning high confidence to incorrect answers. We call this phenomenon Calibration Drift Under Reasoning (CDUR) and study it both theoretically and empirically.\n  We define reasoning budget B and analyze conditions under which Expected Calibration Error ECE(B) follows a non-monotonic pattern: it first decreases as reasoning corrects errors, then increases as longer reasoning produces internally consistent but incorrect explanations. We propose a Hypothesis Lock-In model based on autoregressive generation to ex","title":"Calibration Drift Under Reasoning: How Chain-of-Thought Budgets Induce Overconfidence in Large Language Models","url":"https://arxiv.org/abs/2606.11211","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.11211v1 Announce Type: cross \nAbstract: The ability of large language models (LLMs) to express calibrated uncertainty is important for safe deployment. Chain-of-thought (CoT) reasoning is widely used to improve accuracy and reliability, but its effect on calibration is not fully understood. We show that this picture is incomplete: in some settings, increasing the reasoning budget beyond a task-specific threshold can cause models to become systematically overconfident, assigning high confidence to incorrect answers. We call this phenomenon Calibration Drift Under Reasoning (CDUR) and study it both theoretically and empirically.\n  We define reasoning budget B and analyze conditions under which Expected Calibration Error ECE(B) follows a non-monotonic pattern: it first decreases as reasoning corrects errors, then increases as longer reasoning produces internally consistent but incorrect explanations. We propose a Hypothesis Lock-In model based on autoregressive generation to ex","title":"Calibration Drift Under Reasoning: How Chain-of-Thought Budgets Induce Overconfidence in Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-11T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.11211"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:34042306bd206c3aeb2a85c3927b71e32ab4af205a5c1ac8928550c3094b7107eacb7b90611ecee12d2d55fadcac998ec12b89d1befc530f728c8f781cf5f901","signer":"crovia.substrate","subject":{"observed_at":"2026-06-11T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.11211"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"57920d7b6c5118fb69f122fef46e5aeb1ffae5d6c5eed2cb83a17918e809b252","leaf_index":227371,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"65ff053ea941fee774c256be55c9da7b6e1179348b7bb338c25d6b722001eabf","side":"left"},{"sibling":"06a439be28621d95bfb8720b87f9de874d4c4725fb1434505bf92036e59b39ee","side":"left"},{"sibling":"93ac2d5d29c59579075785decedf1c355146bfc164d0717c2c575a5ddbf42cba","side":"right"},{"sibling":"8520b8eaad68f7bea56d1e6bf78e730873a4ba829017758a65eeff78f6790fc3","side":"left"},{"sibling":"abbc3ebd09d79b8c38c9837ba190a2f1b6c05b8132cf36da695c7537220dc4a4","side":"right"},{"sibling":"af13b87f3f60a427d3566291c1ee132243edd0857ef187e279f49821afe385c3","side":"left"},{"sibling":"98a196a185b8283495ee2af791dddabecad12b9667f3282e20e2ada5f5abd314","side":"right"},{"sibling":"a6d2c01df24bdf78d7a3a20470793a1585a197c5d8c3f07fc4049ed07458246d","side":"right"},{"sibling":"2cfac7f042209c8533c6031bddc4a83bc156e0595f1b28efefbda208904f338a","side":"right"},{"sibling":"04e399458c5b36988cae0bf1c6dbe1b01349003b15cb5aa43f95c55acffe4ec3","side":"right"},{"sibling":"1383228337d54218bd8e5563aebb0b0dfe15c5269e3d5138e64c261d6130a88b","side":"right"},{"sibling":"57cb49c192550231071a0bf53a0821da2f79c585ec6c8d0fc76cebd62ccd78b2","side":"left"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_06817900a0b85758f360d7175d22bee44dfa32ffbd81d7d006b66a67cd657a45"}}