{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_30aa7dbfda541cb03c0461988a78a69462e4c96fa105ac65b361c30e53c811cf","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_30aa7dbfda541cb03c0461988a78a69462e4c96fa105ac65b361c30e53c811cf","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0d1d7edc32eabe8ef0cb803b781d3c8d1e7ca551810e4fddf0fc8300d8c1686f","published":"Wed, 24 Jun 2026 00:00:00 -0400","receipt_hash":"0d1d7edc32eabe8ef0cb803b781d3c8d1e7ca551810e4fddf0fc8300d8c1686f","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0d1d7edc32eabe8ef0cb803b781d3c8d1e7ca551810e4fddf0fc8300d8c1686f","observed_at":"2026-06-24T04:43:17.877668Z","parent_run_hash":"ca17d06d44ba7db934e6f913874699efc608b8f87453f1ac67f52060e620b57c","published":"Wed, 24 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.24281v1 Announce Type: cross \nAbstract: Reasoning language models are increasingly asked not only to answer difficult questions, but also to estimate their likelihood of success. Existing methods typically elicit confidence only once: either before thinking or after answering. We argue that confidence in reasoning models is state-dependent: before thinking, confidence should estimate the chance of the model correctly solving the prompt, while after thinking it should predict whether the realized answer is likely to be correct. This distinction determines the appropriate supervision target: prompt-level success should supervise confidence estimates made after seeing the prompt, while individual answer-level correctness should supervise confidence estimates made after answering. We introduce CALIBER (Calibration Before and After Reasoning), which elicits both estimates and supervises each with the target matched to its information state. Under this unified protocol, CALIBER re","title":"CALIBER: Calibrating Confidence Before and After Reasoning in Language Models","url":"https://arxiv.org/abs/2606.24281","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.24281v1 Announce Type: cross \nAbstract: Reasoning language models are increasingly asked not only to answer difficult questions, but also to estimate their likelihood of success. Existing methods typically elicit confidence only once: either before thinking or after answering. We argue that confidence in reasoning models is state-dependent: before thinking, confidence should estimate the chance of the model correctly solving the prompt, while after thinking it should predict whether the realized answer is likely to be correct. This distinction determines the appropriate supervision target: prompt-level success should supervise confidence estimates made after seeing the prompt, while individual answer-level correctness should supervise confidence estimates made after answering. We introduce CALIBER (Calibration Before and After Reasoning), which elicits both estimates and supervises each with the target matched to its information state. Under this unified protocol, CALIBER re","title":"CALIBER: Calibrating Confidence Before and After Reasoning in Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-24T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.24281"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:2f88b406a3624e4fa373d28a65b35d54f77880e20741e3e61b7dcfb867da4a042c00ea8d225292ff4b571dbe9c1b25a65f14262c652ebebfcfbab5841bfd820f","signer":"crovia.substrate","subject":{"observed_at":"2026-06-24T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.24281"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"a0dba18bef32df72735c34ff024f061c76408aca14bcf50266e76e4dba80d438","leaf_index":244567,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"8c342bc9e20f23a399061b7b749cb16bab816131043ea416fa771c4fe6fe2ce1","side":"left"},{"sibling":"d75cab56ebafadd8389a8c68805829ecdd569c634fe4ef398c3b6b2261e29013","side":"left"},{"sibling":"4f12bc11d384a5978a2399933ac36e2cf8060b7aad78d70631c475c6260d23a4","side":"left"},{"sibling":"1abe02f005cd0d8b0c3cf66807721dec49af0e5e72f31090f215c3e804d8b95b","side":"right"},{"sibling":"92703a887e7f1bf8322f8d44ac0cb59c1842671e6697413180532d2dd09d35d1","side":"left"},{"sibling":"b067534616bb7d7c6554b89a9447081f53dff0fccf744257ad56cf669f711947","side":"right"},{"sibling":"2fb6532259e6852414e5b4ef4dd8ffec5a8a48bee3324303be4c62aabee72329","side":"left"},{"sibling":"d4fcf6bcc7f1fe78997b28df74b23bba21f5b2585d6fe8efb802f0d067291d5c","side":"right"},{"sibling":"0fa23771b702ff726ed1fc5a44f9b416b2a7861c2f957fcac2d95392276d4784","side":"left"},{"sibling":"bc74ebb08462da8a50fc65ea75f8a8a3418d10ebd471d830f1c67f33dd54dfd1","side":"left"},{"sibling":"6dafd355e5d54c60e61c6c02d3842984e234b1f5bca1623fcdd3def7b8931973","side":"right"},{"sibling":"3107b9d4dbf9456a39f99de694a4dd4da2c0600f9f8855f125161335fe8810af","side":"left"},{"sibling":"86118ab4500c3055a2af70062751a960423c464405b18ca1c37411bf0ce3f52e","side":"left"},{"sibling":"3a42039065acac6d3e4088ec61d9c116ecf7a26c7b7163d23da8fd0b3362e038","side":"left"},{"sibling":"c044f2bd864a0e8e8af5a7f6e3124def7fc4b4511b2b166ea8f9de321e8d385e","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":244827,"merkle_root":"274e133c6dfa2781a9cfb85337d01cc6b72688ce5e810149f3183e400ffab136","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260624T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-24T05:37:55Z","sig_algorithm":"ed25519","signature":"22ca3cee4de2447b3d281e30e09fe566461996bb7be4d4465f08a3f4cf59cea58f22683a4aa10ef4d5d17a19b03f4212392bfd26f2b51f289f0cdf1042a03800","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_30aa7dbfda541cb03c0461988a78a69462e4c96fa105ac65b361c30e53c811cf"}}