{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_51f391e81b885d79bcd0e985d5906ea218bf2e25aa61c87b68fe5fe8ab24a7c8","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_51f391e81b885d79bcd0e985d5906ea218bf2e25aa61c87b68fe5fe8ab24a7c8","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f65aa33f985a24f7be962fb3d684e8a201186568c4c6e6eba76e04ff718fcaa2","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"f65aa33f985a24f7be962fb3d684e8a201186568c4c6e6eba76e04ff718fcaa2","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f65aa33f985a24f7be962fb3d684e8a201186568c4c6e6eba76e04ff718fcaa2","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.19868v1 Announce Type: new \nAbstract: Although large language models (LLMs) have shown strong capabilities across a wide range of tasks, their outputs often remain unreliable and may contain hallucinations, making uncertainty estimation (UE) essential for building trustworthy LLMs. In practice, many mainstream LLMs are only accessible through restricted APIs, where internal signals such as logits and hidden states are unavailable, making black-box UE especially important. However, existing work on black-box UE for LLMs remains fragmented in methodology and lacks a unified empirical comparison. To address this gap, we present a systematic review of black-box UE methods and organize them into five categories: verbalization-based, sampling-based, explanation-based, multi-agent, and hybrid methods. We further build a unified evaluation framework and benchmark 24 representative methods across 4 models and 4 dataset settings. Our results show that no single method consistently dom","title":"A Systematic Evaluation of Black-Box Uncertainty Estimation Methods for Large Language Models","url":"https://arxiv.org/abs/2606.19868","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.19868v1 Announce Type: new \nAbstract: Although large language models (LLMs) have shown strong capabilities across a wide range of tasks, their outputs often remain unreliable and may contain hallucinations, making uncertainty estimation (UE) essential for building trustworthy LLMs. In practice, many mainstream LLMs are only accessible through restricted APIs, where internal signals such as logits and hidden states are unavailable, making black-box UE especially important. However, existing work on black-box UE for LLMs remains fragmented in methodology and lacks a unified empirical comparison. To address this gap, we present a systematic review of black-box UE methods and organize them into five categories: verbalization-based, sampling-based, explanation-based, multi-agent, and hybrid methods. We further build a unified evaluation framework and benchmark 24 representative methods across 4 models and 4 dataset settings. Our results show that no single method consistently dom","title":"A Systematic Evaluation of Black-Box Uncertainty Estimation Methods for Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.19868"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:27a7c8e7a99d20e7b451a55bc6c19d214c212bdbbf632a7774ae61df1b28d33a14eafcb184a2d63d192a87394e56e34e40e390c9e7d8ac4a555f146fc870ac04","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.19868"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"3aab6027dbd58d2e2cc8ef3c7622c97c60095f0145e7bf6c1b903a68c184f149","leaf_index":235488,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c637a9a9f43c80858d51f2c0e1568e25501d7bc7c996b1863a4d41a6814dfb7e","side":"right"},{"sibling":"9922a32305848589edb8eb94e5771932e7ce42d3aa96c08d698c3bdc0dc666b0","side":"right"},{"sibling":"3a07bdce1c3823c2a3dd9fe773edf99e7484cfd81190f841d8107bb3e11d91d8","side":"right"},{"sibling":"3c454505fdd0ed9754dc19983be2d7aa9b6178c55b839a01d7b157ce43a4d8b9","side":"right"},{"sibling":"bcd5aa5efe653b441d5f714c138f0ec81e2352e25c05bbf40f11618cd4c25dcb","side":"right"},{"sibling":"9d5b3a47759460fa8b5eceb45de221aafbf0cd60d3ee412f3351ff44394e520f","side":"left"},{"sibling":"696ac6bcf68966ee959b35539f0bed60b3216ccb53620defd29feb8dbcb2ac30","side":"left"},{"sibling":"95ba45f2fea2d822bc863962b7e478ea50086827d90fab88c0da065d454d7ac5","side":"left"},{"sibling":"00f27f149ddd2eb0d2cf60eaed9e1d55c662cf1cf69c95fbea03bba9d35f90c7","side":"left"},{"sibling":"797e0feb6bf826a55956c876711cd24824818c5a84a591f8cc06b95577b3405d","side":"left"},{"sibling":"f049d6e86b410f6f63921a6e3aa684efffa398fd23eed28245c43b156d404c5f","side":"left"},{"sibling":"9ec7f4f84de7e2057a02ac55686567b21acfd5beaecc7a84e9e48f3db17296c2","side":"right"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_51f391e81b885d79bcd0e985d5906ea218bf2e25aa61c87b68fe5fe8ab24a7c8"}}