{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_bf6038bcdb1e23b39b35e41037289c71e5246f028fdbcfc2dcf8e2fb7c196c58","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_bf6038bcdb1e23b39b35e41037289c71e5246f028fdbcfc2dcf8e2fb7c196c58","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"1c695debb108e3d81e1d7f6609a9509bc07dca04d9c20e83e1f138a71552a9ef","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"1c695debb108e3d81e1d7f6609a9509bc07dca04d9c20e83e1f138a71552a9ef","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"1c695debb108e3d81e1d7f6609a9509bc07dca04d9c20e83e1f138a71552a9ef","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.27752v2 Announce Type: replace \nAbstract: LLM confidence calibration is often evaluated by comparing two signals: token-probability scores and verbalized confidence. These signals are sometimes treated as direct readouts of model uncertainty, but their comparison depends on measurement choices that are rarely made explicit. In the main analysis, we hold the verbalized-confidence elicitation fixed: a single prompt template, probability scale, and output format. We then vary the measurement axes that define the verbalized-vs-token comparison: which answer string receives the token-probability score, how that score is read from the answer tokens, and under which conditioning context it is measured. We evaluate this design on four QA benchmarks across three open 7--8B base/Instruct model families, with larger Qwen2.5 variants as same-family robustness checks. The resulting comparison is sensitive to these choices: conditioning context changes the sign or magnitude of the ECE gap","title":"Asking Is Not Enough: Protocol Sensitivity in LLM Confidence Calibration","url":"https://arxiv.org/abs/2605.27752","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.27752v2 Announce Type: replace \nAbstract: LLM confidence calibration is often evaluated by comparing two signals: token-probability scores and verbalized confidence. These signals are sometimes treated as direct readouts of model uncertainty, but their comparison depends on measurement choices that are rarely made explicit. In the main analysis, we hold the verbalized-confidence elicitation fixed: a single prompt template, probability scale, and output format. We then vary the measurement axes that define the verbalized-vs-token comparison: which answer string receives the token-probability score, how that score is read from the answer tokens, and under which conditioning context it is measured. We evaluate this design on four QA benchmarks across three open 7--8B base/Instruct model families, with larger Qwen2.5 variants as same-family robustness checks. The resulting comparison is sensitive to these choices: conditioning context changes the sign or magnitude of the ECE gap","title":"Asking Is Not Enough: Protocol Sensitivity in LLM Confidence Calibration","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.27752"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:72547965f97309ca55eb97768a891e780f3bcecc84cf99a8b88467af923844f7c4b7ee1a3e8a4f7100d34945e625a8d83380b7203702d154d7fcb3710898ba06","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.27752"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8731eaa957cc32e003c9b9fa147e8c552b69059489f9e9f7bda3f7c014e53623","leaf_index":205823,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"0b0e6b64085fe70f57ae688dbfef0ade70bf04606de58139aaf868acdf3e9835","side":"left"},{"sibling":"c2f1aa2c05f19a229aaa15f2fcaa8c89d95303fbfd9af97a08cef55369ba5d5f","side":"left"},{"sibling":"eac6d06b619b92f96b5d7427b8aaf8382581b334f6c2bfd33ba31dc0fdb1b03e","side":"left"},{"sibling":"d1055c1334ff91cd5c60c14a07abe8478c723f36aeeffb7e4f493a94f69351fd","side":"left"},{"sibling":"d1d7b155bb71be1647526836a4b6902904305d382e9d3bf712e766416f695d98","side":"left"},{"sibling":"404bb542eda3ec8629ada5c69a475a2753a9a79f200eca5374244536b7210d7c","side":"left"},{"sibling":"199f39173936d8df4dd145829492a9dceaf32da27938b47dffcfee0b9a047c55","side":"left"},{"sibling":"f90a21898085db59f0f27c56200706376a54d0bca343ed9e47ede5df384cd547","side":"left"},{"sibling":"21332ee1c3955bf1803955bf945ab0966ce5e0aa48d8642320cdbc044a72b935","side":"left"},{"sibling":"6620a5008acf732cd3e57b0f2d1437293e88366c2e5b175ebeb7d796fc1e0c62","side":"left"},{"sibling":"a82575bfb494af7afcc13aaae718afa6f74030f09d71b02819ea25efd4186fc4","side":"right"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_bf6038bcdb1e23b39b35e41037289c71e5246f028fdbcfc2dcf8e2fb7c196c58"}}