{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_acd0ca84eb22cad5ab783463b984110a1e3d7f80dcfb8f86a836ec0e78e17a7f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_acd0ca84eb22cad5ab783463b984110a1e3d7f80dcfb8f86a836ec0e78e17a7f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"23f299f8effe22e4a44cbb21fad92b66b622cd25e4dfa98336b195f3fdf410b1","published":"Thu, 28 May 2026 00:00:00 -0400","receipt_hash":"23f299f8effe22e4a44cbb21fad92b66b622cd25e4dfa98336b195f3fdf410b1","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"23f299f8effe22e4a44cbb21fad92b66b622cd25e4dfa98336b195f3fdf410b1","observed_at":"2026-05-28T04:43:38.862500Z","parent_run_hash":"58f8b4a134069e0a15ea3949252489597eb86dd27c9ca3fb15c6fb838ce49ef3","published":"Thu, 28 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.28500v1 Announce Type: cross \nAbstract: Large language models have shown impressive capabilities in code generation, yet they often produce functionally incorrect code. Uncertainty quantification (UQ) methods have emerged as a promising approach for detecting hallucinations in natural language generation, but their effectiveness for code generation tasks remains underexplored. We systematically evaluate how UQ techniques transfer to code generation across three programming languages, five LLMs, and over 1,700 problems. We find that some token-probability-based methods generalize effectively without modification, while sampling-based methods relying on natural language inference (NLI) fail because NLI models cannot distinguish functionally different code, causing most responses to collapse into a single semantic cluster. To address this, we introduce functional equivalence methods, a family of code-specific methods that replace NLI-based semantic equivalence with an LLM-based","title":"Functional Entropy: Predicting Functional Correctness in LLM-Generated Code with Uncertainty Quantification","url":"https://arxiv.org/abs/2605.28500","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.28500v1 Announce Type: cross \nAbstract: Large language models have shown impressive capabilities in code generation, yet they often produce functionally incorrect code. Uncertainty quantification (UQ) methods have emerged as a promising approach for detecting hallucinations in natural language generation, but their effectiveness for code generation tasks remains underexplored. We systematically evaluate how UQ techniques transfer to code generation across three programming languages, five LLMs, and over 1,700 problems. We find that some token-probability-based methods generalize effectively without modification, while sampling-based methods relying on natural language inference (NLI) fail because NLI models cannot distinguish functionally different code, causing most responses to collapse into a single semantic cluster. To address this, we introduce functional equivalence methods, a family of code-specific methods that replace NLI-based semantic equivalence with an LLM-based","title":"Functional Entropy: Predicting Functional Correctness in LLM-Generated Code with Uncertainty Quantification","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-28T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.28500"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:58a1b31cb54f086625037304056982aac846c8e55149e62a0a0cddacdc2a89c2b3bd783dca926dcebd35777cc8bd65caeede72c509060fb7780a174f82e38707","signer":"crovia.substrate","subject":{"observed_at":"2026-05-28T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.28500"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7fcc3854c892a35e222fd78c6111d7f1905a39e896eb9f3f68983edf68942651","leaf_index":155901,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"419b1991b50fbb555f8a0d4ff596571449c3c7161e5ce6cf22a0427d29cf55d3","side":"left"},{"sibling":"0d7ac3a14d83d8f96251ea55f9fe167ad7a14368e5ad47abcff9e9f2ce820dd8","side":"right"},{"sibling":"e7bf5c8952b8015bb284fa9d1c4f7f7f6ab5e7fed56f7176777668dd950a67c9","side":"left"},{"sibling":"f2dc462c77b4a036179dc5a996f01b610b388132edc9163b2a638f5a1d3620c7","side":"left"},{"sibling":"ded3dd76669a4119c34d9bb6ec7427c1230c38b517c47b29332f746bf562cfaf","side":"left"},{"sibling":"d6e9eb8cef332764b9d8d1d9230e18ceb0aeab2a041b3f73421557267f9a5193","side":"left"},{"sibling":"23545a0a9721502eebf2f3a4b30acec3fc420534240d53ac993fddfb75131c5e","side":"left"},{"sibling":"753018fa645affe1bfe31d1c496292a6e72167116d741961e921b85a693e4153","side":"left"},{"sibling":"0adb35dd3ad9bfc4a781c02606acf0a3ce1e28e1a88066209cc1a35814e790ad","side":"right"},{"sibling":"f292d3278e493ec60902181b8c0bd5c89c0a1168ef928222fe6161287982f7f0","side":"right"},{"sibling":"2209295faf1a5bf51c97c6fd5a839a8181a4ea44f490420381530f35df7d9b2f","side":"right"},{"sibling":"5784576a15214ea9fc3569e6e1cff1ef443c0b1fc0d036028489088af089de27","side":"right"},{"sibling":"311772ec218efcb2da5a337f9e9f042fe1cc0028643adb0a354787e4ea7911b7","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"7b6f0bea4291a5e63574dfca9aa0f9756f450c9478c3d07474d39a7ababb51f9","side":"right"},{"sibling":"1d39fe14b21e2ebbfb87e882423b24ee9469eae1e4c77af5b799ac4db9537467","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":156177,"merkle_root":"c5705a0243d16afd8b1ebfd731b7aa304079c442c2a7906493c5bbed374c69ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260528T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-28T05:37:36Z","sig_algorithm":"ed25519","signature":"f087e13febc8bb6a2e0812610de64cebc65be92915518d9c4b230799c3b161839b04c4eb1b02741f938f35545a76ab76555b04c782bdc2f9a44852d171d65909","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_acd0ca84eb22cad5ab783463b984110a1e3d7f80dcfb8f86a836ec0e78e17a7f"}}