{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_738f7a0eb50f6f445773ac9fbff81f7a724f34c0f48314bd3b17b28a6118ab2d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_738f7a0eb50f6f445773ac9fbff81f7a724f34c0f48314bd3b17b28a6118ab2d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"d8658354fc395e2c12d84a5529dfe039802102478bc54d163c15e0b88c03c295","published":"Wed, 24 Jun 2026 00:00:00 -0400","receipt_hash":"d8658354fc395e2c12d84a5529dfe039802102478bc54d163c15e0b88c03c295","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"d8658354fc395e2c12d84a5529dfe039802102478bc54d163c15e0b88c03c295","observed_at":"2026-06-24T04:43:17.877668Z","parent_run_hash":"ca17d06d44ba7db934e6f913874699efc608b8f87453f1ac67f52060e620b57c","published":"Wed, 24 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2601.22548v4 Announce Type: replace-cross \nAbstract: Recent research has shown that large language models (LLMs) favor their own outputs when acting as judges, undermining the integrity of automated post-training and evaluation workflows. However, it is difficult to disentangle which behaviors are explained by narcissism versus experimental confounds. Specifically, LLM evaluators may deliver self-preferring verdicts when comparing responses to questions they fail on; these verdicts may not depend on the identity of the author, but on evaluator quality. We correct this by directly comparing the judge's voting distribution in cases where it evaluates itself versus another model. This evaluator quality baseline reveals that only 51% of examples in previous findings retain statistical significance against this null hypothesis, covering 89.6% of total self-preference probability mass. Finally, we compare the entropy of voting distributions, suggesting uncertainty-driven overlap, and s","title":"Are LLM Evaluators Really Narcissists? Sanity Checking Self-Preference Evaluations","url":"https://arxiv.org/abs/2601.22548","vendor":"arxiv_cs_ai"},"summary":"arXiv:2601.22548v4 Announce Type: replace-cross \nAbstract: Recent research has shown that large language models (LLMs) favor their own outputs when acting as judges, undermining the integrity of automated post-training and evaluation workflows. However, it is difficult to disentangle which behaviors are explained by narcissism versus experimental confounds. Specifically, LLM evaluators may deliver self-preferring verdicts when comparing responses to questions they fail on; these verdicts may not depend on the identity of the author, but on evaluator quality. We correct this by directly comparing the judge's voting distribution in cases where it evaluates itself versus another model. This evaluator quality baseline reveals that only 51% of examples in previous findings retain statistical significance against this null hypothesis, covering 89.6% of total self-preference probability mass. Finally, we compare the entropy of voting distributions, suggesting uncertainty-driven overlap, and s","title":"Are LLM Evaluators Really Narcissists? Sanity Checking Self-Preference Evaluations","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-24T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2601.22548"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:c9d47f4b638578aef16950f0d4d3359890a7471aae17cb6697c33045d62e1b249e148e091c8ea096931cc5f431a491af66487b07fe64efb2a5e168f7e185ee00","signer":"crovia.substrate","subject":{"observed_at":"2026-06-24T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2601.22548"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"02ae1a367b4f42cde77ae57011486b67fb9bd2e5e7488b421eefc96764c116fe","leaf_index":244661,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"0358718e6f0effc23312d32e2ced43947b733f9a2c6bfc1ad438a12ae4c167f8","side":"left"},{"sibling":"c8ddbe7246a9dfbadce928595c4ebbd5af93f28d9b5755c4cba7ecf01a16a1dc","side":"right"},{"sibling":"4bff68e29b3b39dc80eb5ebb33a34c2c1985c34d2fe116bb37da58364e557a30","side":"left"},{"sibling":"805dfe86c51f01bc8b191fd723fb4b847e02c53e171f9bfd038e1a1d0e5b0508","side":"right"},{"sibling":"2586872d8d2b7bfaf17cf5d7615fb3267f6eea77c1691fe7be6a420126730851","side":"left"},{"sibling":"10a3b1b138de9ae6fc16e343494741456778bc5b35b82a012de1660de957a2a5","side":"left"},{"sibling":"39f1beec9e2d4b8c80e6bc14c5c36212db399f0b56af124c0b694f66d90b987b","side":"right"},{"sibling":"6a0452a49c95630662e0ea8383207f49cc3a62727b575152a766fcfccfdc3998","side":"left"},{"sibling":"0fa23771b702ff726ed1fc5a44f9b416b2a7861c2f957fcac2d95392276d4784","side":"left"},{"sibling":"bc74ebb08462da8a50fc65ea75f8a8a3418d10ebd471d830f1c67f33dd54dfd1","side":"left"},{"sibling":"6dafd355e5d54c60e61c6c02d3842984e234b1f5bca1623fcdd3def7b8931973","side":"right"},{"sibling":"3107b9d4dbf9456a39f99de694a4dd4da2c0600f9f8855f125161335fe8810af","side":"left"},{"sibling":"86118ab4500c3055a2af70062751a960423c464405b18ca1c37411bf0ce3f52e","side":"left"},{"sibling":"3a42039065acac6d3e4088ec61d9c116ecf7a26c7b7163d23da8fd0b3362e038","side":"left"},{"sibling":"c044f2bd864a0e8e8af5a7f6e3124def7fc4b4511b2b166ea8f9de321e8d385e","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":244827,"merkle_root":"274e133c6dfa2781a9cfb85337d01cc6b72688ce5e810149f3183e400ffab136","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260624T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-24T05:37:55Z","sig_algorithm":"ed25519","signature":"22ca3cee4de2447b3d281e30e09fe566461996bb7be4d4465f08a3f4cf59cea58f22683a4aa10ef4d5d17a19b03f4212392bfd26f2b51f289f0cdf1042a03800","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_738f7a0eb50f6f445773ac9fbff81f7a724f34c0f48314bd3b17b28a6118ab2d"}}