{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_a3e9d7a7e36d306412bc921ec2172340a397f04d83621b736380515bdef95590","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_a3e9d7a7e36d306412bc921ec2172340a397f04d83621b736380515bdef95590","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"fd2cf3cecb4d6a78aa1502da984f39135c9159b4cbf005971a2a9386b4b5a26d","published":"Fri, 24 Jul 2026 00:00:00 -0400","receipt_hash":"fd2cf3cecb4d6a78aa1502da984f39135c9159b4cbf005971a2a9386b4b5a26d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"fd2cf3cecb4d6a78aa1502da984f39135c9159b4cbf005971a2a9386b4b5a26d","observed_at":"2026-07-24T04:43:08.456021Z","parent_run_hash":"b018378f86139a28e6209ec008b31c1282cd1b5c1632dbd43b054c17aa88ab96","published":"Fri, 24 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.20526v1 Announce Type: new \nAbstract: Large language models (LLMs) are increasingly deployed in settings where fluent but incorrect answers can be costly. In these settings, accuracy alone is insufficient: models must also know when they are likely to be wrong. We present ConfidenceBench, a calibration benchmark that evaluates verbalized confidence estimates in 15 frontier LLMs using the Brier score, a proper scoring rule that incentivises truthful probability reporting. Confidence is elicited via prompting, requiring no access to model logits and making the framework applicable to both closed-source and open-source systems. The benchmark comprises 200 private multiple-choice questions across four categories: spatial reasoning, high-precision mathematics, word lookup, and unknowable questions. Across three independent runs, Claude Opus 4.6 and Gemini 3.1 Pro Preview achieve the lowest Brier scores, both reported as 0.103. Both substantially outperform the calibrated-random b","title":"ConfidenceBench: Evaluating Confidence Calibration in Large Language Models","url":"https://arxiv.org/abs/2607.20526","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.20526v1 Announce Type: new \nAbstract: Large language models (LLMs) are increasingly deployed in settings where fluent but incorrect answers can be costly. In these settings, accuracy alone is insufficient: models must also know when they are likely to be wrong. We present ConfidenceBench, a calibration benchmark that evaluates verbalized confidence estimates in 15 frontier LLMs using the Brier score, a proper scoring rule that incentivises truthful probability reporting. Confidence is elicited via prompting, requiring no access to model logits and making the framework applicable to both closed-source and open-source systems. The benchmark comprises 200 private multiple-choice questions across four categories: spatial reasoning, high-precision mathematics, word lookup, and unknowable questions. Across three independent runs, Claude Opus 4.6 and Gemini 3.1 Pro Preview achieve the lowest Brier scores, both reported as 0.103. Both substantially outperform the calibrated-random b","title":"ConfidenceBench: Evaluating Confidence Calibration in Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-24T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.20526"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:dd6a0bad5cca24f6ee762d6ef93a5164ae5710b4a0b83fc014da52ec22414c82ad63d303b65b0702fb7effb66c4bb4289fc500424bbb17a20b124c2b5d83b506","signer":"crovia.substrate","subject":{"observed_at":"2026-07-24T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.20526"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"6fb54b62ae9d2542c3fcd240651d3d6d0b9d201a8af30788ae61dfe1b03db322","leaf_index":346993,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fef6f5ac70acff5063e7448bd3928b1a62a44fb9751d684a7ca1e5ce5bf2010a","side":"left"},{"sibling":"a9e955bfb2706f5548354af8bb831b5b76c0b3f13548674ab827f989c37848a5","side":"right"},{"sibling":"3c37b17b72c2a45170b8944b9a7fa3511d5db21e46a573f5371f8b9f90225e8e","side":"right"},{"sibling":"5728c1cfa777d07537145dd4a42a073e02842e5013274c3aea6744d547a7b53b","side":"right"},{"sibling":"95da4ad14827dc8554eb82f12569537c1ac46e8a84fbd2e08d4f0ecf82fb98db","side":"left"},{"sibling":"2d8571a85ef8db282687baf52c678be7c1e616f7a8b798c9f0b7c1f94ce5acd7","side":"left"},{"sibling":"6149df3d1da13abe8a81da069f7bd406a0badf8b62cfc2d62ba645f403037400","side":"left"},{"sibling":"ffd3a55995db3d94b2e5c6432ae18b8c5c13fa05a460ebe646d97e4cfd4c196c","side":"right"},{"sibling":"c212fcc83c532c0321804b72fe72b546946bb055d3d482aab19c43f5bebfbf3f","side":"left"},{"sibling":"da189c159d0789c2229cf3731890cc753832dab1aa83e4bbf0fac01955c22cd8","side":"left"},{"sibling":"3a02ed8ea41a09957282e4e27db74ed88cd675156461abb02acb28e2cd257e63","side":"right"},{"sibling":"d3139af8c5ce235438e1c69e4b7afa44ba09129fd86674968434f23e546f423e","side":"left"},{"sibling":"cb89775a838ee16d10fc8da3213420c2012b4d96e8d55cd49939b0887a4b92d3","side":"right"},{"sibling":"252d30ea8052c3bb6b40bc5cc29fc9b9725343d212f84c08fbbae4215a125b00","side":"right"},{"sibling":"f3e45bceed774d2402fa45d41ff5190f295823bd2f216eb90157884150034693","side":"left"},{"sibling":"3cfa2102c0224815c6f3bf73e6710e24103f43f7bf5da1ca2abad1416d9c0890","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"e871fd7edf9b2ad89bce1609a028f5225eea4d14372169bac242420830f86530","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":347413,"merkle_root":"9efe042c5dd6583dfd3b6a58fbfc289807f60bcf2bd2927f10488a54a8ba11fc","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260724T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-24T05:38:42Z","sig_algorithm":"ed25519","signature":"8633c55f558d42994850505218b2862c6134bad2b1c80d4b80736c2fd3ea7a19690ca3498727a8adbdc47791176128a8d64bef0883b888db809f0477355bd00c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_a3e9d7a7e36d306412bc921ec2172340a397f04d83621b736380515bdef95590"}}