{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_0d855ee17d06ea3f7634665b3f39fbe157fcc93d8eeb6614b0e698aa312b1a66","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_0d855ee17d06ea3f7634665b3f39fbe157fcc93d8eeb6614b0e698aa312b1a66","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"4c121f76e8dea4179ba53623f4e64b74727a91419fc427b2fbebdda237fed66f","published":"Fri, 15 May 2026 00:00:00 -0400","receipt_hash":"4c121f76e8dea4179ba53623f4e64b74727a91419fc427b2fbebdda237fed66f","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"4c121f76e8dea4179ba53623f4e64b74727a91419fc427b2fbebdda237fed66f","observed_at":"2026-05-15T04:43:17.611638Z","parent_run_hash":"5c64f85625fabd323e9c4a1cf068c012fb88a248deda9a9ac702fb2f9799f2e5","published":"Fri, 15 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.22891v3 Announce Type: replace-cross \nAbstract: LLM-as-a-Judge has become a dominant approach in automated evaluation systems, playing critical roles in model alignment, leaderboard construction, quality control, and so on. However, the scalability and trustworthiness of this approach can be substantially distorted by Self-Preference Bias (SPB), which is a directional evaluative deviation in which LLMs systematically favor or disfavor their own generated outputs during evaluation. Existing measurements rely on costly human annotations and conflate generative capability with evaluative stance, and thus are impractical for large-scale deployment in real-world systems. To address this issue, we introduce a fully automated framework to quantifying and mitigating SPB, which constructs equal-quality pairs of responses with negligible quality differences, enabling statistical disentanglement of discriminability from bias propensity without human gold standards. Empirical analysis a","title":"Quantifying and Mitigating Self-Preference Bias of LLM Judges","url":"https://arxiv.org/abs/2604.22891","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.22891v3 Announce Type: replace-cross \nAbstract: LLM-as-a-Judge has become a dominant approach in automated evaluation systems, playing critical roles in model alignment, leaderboard construction, quality control, and so on. However, the scalability and trustworthiness of this approach can be substantially distorted by Self-Preference Bias (SPB), which is a directional evaluative deviation in which LLMs systematically favor or disfavor their own generated outputs during evaluation. Existing measurements rely on costly human annotations and conflate generative capability with evaluative stance, and thus are impractical for large-scale deployment in real-world systems. To address this issue, we introduce a fully automated framework to quantifying and mitigating SPB, which constructs equal-quality pairs of responses with negligible quality differences, enabling statistical disentanglement of discriminability from bias propensity without human gold standards. Empirical analysis a","title":"Quantifying and Mitigating Self-Preference Bias of LLM Judges","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-15T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.22891"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:49c0e4f462632f8deb9cce2f91bef9208c90853b6986db0c17e1c8a3b8fc4b2bc6682f3756bf5a232876f30d8267c04e01100c69e522a2fc23619d7315b2d402","signer":"crovia.substrate","subject":{"observed_at":"2026-05-15T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.22891"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d2ebad65aa2d5c304875426be230f49a1f0d5a2026c5751aeec89d83ddd261f1","leaf_index":134800,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fd713c635a84307d2e738b8ada94443695f736954265f76853e52d97338f959f","side":"right"},{"sibling":"14de18b300e806f7d0eee9263ac5f3fbffe863f59fdb27744ccbf912c1c77f48","side":"right"},{"sibling":"227889165d75ae840a32a752c9590ff7f1cf94d6fbf8cf97fd49864149575de3","side":"right"},{"sibling":"686d3031b0a2fe1016ed013e5017884927672094c4da2307f3ac7a34c05c10c1","side":"right"},{"sibling":"4bf0085792aaba9c39d4ba78c58255609bce019f21775220a541d055902f9515","side":"left"},{"sibling":"5fc2935f973135027dfec44abb4984cf3256aee30f1888041f5f4c94633bf55a","side":"right"},{"sibling":"7fb440277b43a0a448cbfa0e516ee5aa0504495adea486e423b8bbf045b1f798","side":"right"},{"sibling":"c007bf5d6c3a3a484205f30ad38ce03dff757769991af941772ae831255ede50","side":"left"},{"sibling":"a17a0eaa87574a308e2c02cf125072b9e69566b9e8d6d2e64a02880192b885d5","side":"right"},{"sibling":"6bd0475fd3a73322a4f73095c96b072de89e462ce5839161aa552e0208619bce","side":"left"},{"sibling":"36672459e5ed50c64ee1842b69cb6d2eb682c2a04844555be8d124257571994a","side":"left"},{"sibling":"727783827652adfa99c455bd80a01bfb33836228e51068b4f654ef3da468ca69","side":"left"},{"sibling":"623194cd30880ed223e306737fdb111aa0d781751bfc47553c404a6af6aad2c4","side":"right"},{"sibling":"fc8f53ed42756907fb79ee19a4ed09f72c560e5302b3d98198b96bf1da635a4a","side":"right"},{"sibling":"d6607539da7ba39ec68be2d12f27ed6768766c745e3120fd915f88c5e288e07c","side":"right"},{"sibling":"b63408a424d27cd6a75e0fb155e69a58a328f41e9cb9dba1eddef9a5289cc7fd","side":"right"},{"sibling":"356fb36a4e188f03d7a05c54cd8789bdd40eda454b9bc9560f667acc08e6c4e0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":134886,"merkle_root":"c6c7ae28c065bced89e7f844216b073f1a7cc4b378db0d41a98bcd21b28066db","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260515T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-15T05:37:26Z","sig_algorithm":"ed25519","signature":"5a3978c26017daf4104adbb3e1c3099c5750157acbfc7242ece1815dc6740fe08a690291ce0afe42011e20cc565b5ebe64ec016bf658b5bab563a63337985c05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_0d855ee17d06ea3f7634665b3f39fbe157fcc93d8eeb6614b0e698aa312b1a66"}}