{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_13d322035bb6d6d1fe35e7e559050aac49fc9c8262e25bb992de7e3cd00d1a7c","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_13d322035bb6d6d1fe35e7e559050aac49fc9c8262e25bb992de7e3cd00d1a7c","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"c57f40f0f6e57c4922416527341278a952873483baf5c0d4115cbcd61d53e417","published":"Mon, 20 Jul 2026 00:00:00 -0400","receipt_hash":"c57f40f0f6e57c4922416527341278a952873483baf5c0d4115cbcd61d53e417","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"c57f40f0f6e57c4922416527341278a952873483baf5c0d4115cbcd61d53e417","observed_at":"2026-07-20T04:43:09.641409Z","parent_run_hash":"0fd83663f0f57da59b26313ca1a35283a3e9f06e3165d3e143d26f7174743aca","published":"Mon, 20 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.15190v2 Announce Type: replace \nAbstract: AI benchmarks increasingly leverage item-level statistical models, particularly item response theory (IRT), to estimate model capabilities, rank systems, select informative examples, and diagnose benchmark quality. However, AI benchmark data often departs from the data regime of human testing, for which standard IRT estimation tools were originally developed: benchmarks typically involve fewer evaluated models, far more items, and capability distributions that may be skewed, clustered, or multimodal. We examine how these regime mismatches challenge the reliability of IRT modeling for AI evaluation. Using item parameters and capability distributions derived from six widely used LLM benchmarks, we simulate response matrices under three common IRT models and compare four estimation tools used in recent benchmark studies: marginal maximum likelihood, Markov chain Monte Carlo, variational inference, and a neural pseudo-Siamese estimator. ","title":"Can We Trust Item Response Theory for AI Evaluation?","url":"https://arxiv.org/abs/2607.15190","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.15190v2 Announce Type: replace \nAbstract: AI benchmarks increasingly leverage item-level statistical models, particularly item response theory (IRT), to estimate model capabilities, rank systems, select informative examples, and diagnose benchmark quality. However, AI benchmark data often departs from the data regime of human testing, for which standard IRT estimation tools were originally developed: benchmarks typically involve fewer evaluated models, far more items, and capability distributions that may be skewed, clustered, or multimodal. We examine how these regime mismatches challenge the reliability of IRT modeling for AI evaluation. Using item parameters and capability distributions derived from six widely used LLM benchmarks, we simulate response matrices under three common IRT models and compare four estimation tools used in recent benchmark studies: marginal maximum likelihood, Markov chain Monte Carlo, variational inference, and a neural pseudo-Siamese estimator. ","title":"Can We Trust Item Response Theory for AI Evaluation?","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-20T04:43:09Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.15190"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:69447e35a71ead138dd95677ef7f8e1fe55a24418cacdf32a1cd8c5d4aada056083f8ba0c032f25e3d623544f20e0f5baf884afa8bc5da2ba2256770a96bc404","signer":"crovia.substrate","subject":{"observed_at":"2026-07-20T04:43:09Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.15190"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7e10e732eb0f925668127666c97d42f09d41fcb86a5247deb45283db239cea79","leaf_index":333359,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5bb26e9d2a5957db5e6e0706382af1c1e80c25c6373a6a336e7e1a7f3dcd6a2d","side":"left"},{"sibling":"347d414cdd903271986d3749c9170de86f012185ca7bbf74406ac97aa7f5bca0","side":"left"},{"sibling":"b002e52d982604ce687d2c8cc5ff2b06f28544b9df8827d4487b9b27d255bac2","side":"left"},{"sibling":"bc24bcef91de9358d2e0a8b094cca7d71d302b59d828477c03becee376a3bab8","side":"left"},{"sibling":"7a7893cc8f38a5f9a49eedec199c3ec5e1a71385c85563ac0cea3cb31a258d29","side":"right"},{"sibling":"0d19791e9aa074b8130eeb2e01249774582a5873ad2fb1aeac0e5267891a9a37","side":"left"},{"sibling":"335e5d86a4ea00f87721afcb0e730f04bbef2c17cd5489f5b8263047ffc4ccd3","side":"right"},{"sibling":"5f23b1a0a67d8fa1aa690680510620f249f8d6683d46c202fca306510fbe398e","side":"right"},{"sibling":"f720760992870795e6d2b913ad9162f9e144b821ea97e4dada744d0cac06e93b","side":"right"},{"sibling":"45c0e4431502711514503abd48e8b3d34ee2f4bffa994c883aaf2adecd0ad8e9","side":"left"},{"sibling":"dedd2da92d9447ddf1b1db68fe20109a426ed18359861ed746907ac820021a8d","side":"left"},{"sibling":"a1c43cc7cd9c775fac33940ee5124aece01596733f003fc53743f43483f9f597","side":"right"},{"sibling":"b5ad3eafd7eeb74c063261356fdd9bf6059ee6d0bf1e3c70e60731b394a5536e","side":"left"},{"sibling":"93e399d152203db688c6a5a58d25131205603504f5b79123a1f2b5a5ed9c1e54","side":"right"},{"sibling":"b6e0cad7f6eb9107f0edd276f1a9942635d8cd6d60d2a97e7daac08b110dc209","side":"right"},{"sibling":"80ec062e7e625dc3f9bb5865cb5198696bbec2608e48abae5670677b90695899","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"0aced6f0c9dec3e6cc9e89b68b70f5f8ce7e1eb13606d92db1917b76e57393c7","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":333540,"merkle_root":"ee60f62b8a724dd9bde638d638caf32cefec4440f832018b457ff47a0ec56a8c","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260720T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-20T05:38:36Z","sig_algorithm":"ed25519","signature":"82a3787e628bfab19c377d875220e1aaedfc498736c545f4708a1b887e8398afdf306995994493a36c864ff7139a2d436b905ce7081aaa540789aa4f707dc800","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_13d322035bb6d6d1fe35e7e559050aac49fc9c8262e25bb992de7e3cd00d1a7c"}}