{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1401ae89b559979daca4bbcab4592057cf5d91d09ed2b5282e051d263cf31912","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1401ae89b559979daca4bbcab4592057cf5d91d09ed2b5282e051d263cf31912","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"83ec7f1a4143aed10e79ede5c693f3857370d3ba93ebad2ce3968a0894a5de3c","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"83ec7f1a4143aed10e79ede5c693f3857370d3ba93ebad2ce3968a0894a5de3c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"83ec7f1a4143aed10e79ede5c693f3857370d3ba93ebad2ce3968a0894a5de3c","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.19714v1 Announce Type: cross \nAbstract: Large language models (LLMs) are increasingly used as judges for open-ended generation, as large-scale human evaluation is often expensive and difficult to scale, yet their preferences remain imperfect proxies for human judgment. Existing auditing pipelines often assume that a reliable subset of examples or clean supervision signals are available beforehand, for example from human annotation, heuristic filtering, or the outputs of strong judges. In LLM evaluation, this assumption is fragile: the initial split may inherit judge bias, while human verification is typically too scarce to define stable groups at scale. We propose AURA, an adaptive uncertainty--aware refinement framework for auditing pairwise LLM--as--a--judge decisions under selected human verification. AURA iteratively learns a human-consistency signal, propagates reliable evidence, and prioritizes uncertain comparisons for human review. The key idea is to treat trust in a","title":"AURA: Adaptive Uncertainty-aware Refinement for LLM-as-a-Judge Auditing","url":"https://arxiv.org/abs/2606.19714","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.19714v1 Announce Type: cross \nAbstract: Large language models (LLMs) are increasingly used as judges for open-ended generation, as large-scale human evaluation is often expensive and difficult to scale, yet their preferences remain imperfect proxies for human judgment. Existing auditing pipelines often assume that a reliable subset of examples or clean supervision signals are available beforehand, for example from human annotation, heuristic filtering, or the outputs of strong judges. In LLM evaluation, this assumption is fragile: the initial split may inherit judge bias, while human verification is typically too scarce to define stable groups at scale. We propose AURA, an adaptive uncertainty--aware refinement framework for auditing pairwise LLM--as--a--judge decisions under selected human verification. AURA iteratively learns a human-consistency signal, propagates reliable evidence, and prioritizes uncertain comparisons for human review. The key idea is to treat trust in a","title":"AURA: Adaptive Uncertainty-aware Refinement for LLM-as-a-Judge Auditing","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.19714"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b11ec8345d613f5fdef61f6e2dea54827343e342f953d6720e2116c61148cc3881ab3314f3aeb92bf34b641cc5a113b1b741ccf90e3a042164592ed10e52150e","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.19714"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c712446972bfa53184c878234ec20c7d8a619b0ca118807e6d2ff77a3c156814","leaf_index":235589,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f8ad53c5af9febae59e20d2552122b42b552d96f84a8c8290d91b6eb488fd7d2","side":"left"},{"sibling":"eb526127df3301159dbb436e0b17a58d5757a934a1577f9049780f716fa63802","side":"right"},{"sibling":"67cc623be383bc349ade513e875a8c19510be18d002d4ab8a8c7e009177ec9bf","side":"left"},{"sibling":"29e8615334bb31e06627b4abdc0e4f3cb639b021f43efc26e864c6566823569f","side":"right"},{"sibling":"14ec5cb3f5183203f0fbe29024e831ad1077f4ca7b3299d4c1fab3b9632b9194","side":"right"},{"sibling":"46d2f9215acf9e742e7178286c5cebf1602211a137b69f030149311b5b21d03c","side":"right"},{"sibling":"eb70d4eaf5e9558deed789daa5addfb31fc38e81a3f06b590069834b28d61e1c","side":"left"},{"sibling":"c0ab0c7dd98a2c17ee965cb19831d49d3a82a2ba1812d8a677f78274ab6f7c98","side":"right"},{"sibling":"aa68ebe8f5e8e96388fc8d1af3aa08be7ccd27ab4cebcf5913560c48d877bc27","side":"right"},{"sibling":"8253d44cf1ed30d3ab19c2b339fb4000a1fa173182c65390e9e8dabf8173b9e9","side":"right"},{"sibling":"e2bf9b60400244c698c0196109f54323457abc5b64dee08ec33ab14cc4faaef7","side":"right"},{"sibling":"86664e7f68ba08b8dfcf77dda51a4dfa7fcfc986d4ad7c704ffb71b669202da7","side":"left"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1401ae89b559979daca4bbcab4592057cf5d91d09ed2b5282e051d263cf31912"}}