{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_da55e2a150ae77dd9c4cf13b13f44f065232322fc7e55d1890daafa5b006164a","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_da55e2a150ae77dd9c4cf13b13f44f065232322fc7e55d1890daafa5b006164a","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"dc27331564ec4f19dd0f6aff6fa2cb973b879cc1e6049a5dc13e35bbb76ed7af","published":"Fri, 26 Jun 2026 00:00:00 -0400","receipt_hash":"dc27331564ec4f19dd0f6aff6fa2cb973b879cc1e6049a5dc13e35bbb76ed7af","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"dc27331564ec4f19dd0f6aff6fa2cb973b879cc1e6049a5dc13e35bbb76ed7af","observed_at":"2026-06-26T04:43:58.168958Z","parent_run_hash":"9459505a803125e4b968df08c74ed0054a2aafd44e4e1a036e3b0709a8a65cb4","published":"Fri, 26 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.26836v1 Announce Type: new \nAbstract: Existing benchmarks typically report accuracy for a single model on a single run. This systematically understates real-world LLM capabilities, particularly under heterogeneous data distributions: (i) different models get different questions correct according to their specializations, and (ii) given a budget, multiple generations can be sampled and selectively retained. To quantify this gap, we introduce the Capability Frontier: a Pareto frontier over a set of models that characterizes the best achievable performance at each cost level under optimal selection across models and generations (i.e., via an oracle). Our construction corrects for two opposing biases: underestimation from single-model evaluation and overestimation from taking maxima over noisy samples. We study 21 LLMs across 16 widely used benchmarks spanning coding, reasoning, medicine, factuality, instruction following, and agentic tasks, comparing Capability Frontier perform","title":"The Capability Frontier: Benchmarks Miss 82% of Model Performance","url":"https://arxiv.org/abs/2606.26836","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.26836v1 Announce Type: new \nAbstract: Existing benchmarks typically report accuracy for a single model on a single run. This systematically understates real-world LLM capabilities, particularly under heterogeneous data distributions: (i) different models get different questions correct according to their specializations, and (ii) given a budget, multiple generations can be sampled and selectively retained. To quantify this gap, we introduce the Capability Frontier: a Pareto frontier over a set of models that characterizes the best achievable performance at each cost level under optimal selection across models and generations (i.e., via an oracle). Our construction corrects for two opposing biases: underestimation from single-model evaluation and overestimation from taking maxima over noisy samples. We study 21 LLMs across 16 widely used benchmarks spanning coding, reasoning, medicine, factuality, instruction following, and agentic tasks, comparing Capability Frontier perform","title":"The Capability Frontier: Benchmarks Miss 82% of Model Performance","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-26T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.26836"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:f5a87e4d3d005825d46a0872bb0197e43b260ccb287a75fde534a15d3c3f5648db5ae51fd3814f8f4a4d755680f13ddcfedc0c21c3841fead99cc7f994e0ac09","signer":"crovia.substrate","subject":{"observed_at":"2026-06-26T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.26836"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"ef3c951bc48582dc7233f792338ffbe6b48397421d0a46d17d4e5ac4272ea70f","leaf_index":251036,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7a6da8d7991c9087a39e521cf6b78b9f1be55f1a4b04f7cced46d4be19d21bd3","side":"right"},{"sibling":"cc5c9c67524eb6bc7525dff28dae885d5afc0394e0a48938448f3be3a773e9da","side":"right"},{"sibling":"ce6bbd5752425196b9664597f929a4b70a480b4a0bab4b4ae1269b28876115de","side":"left"},{"sibling":"afc3689db51a0a9b33a27c5cca6b3802cd8856c0c7d33041d0bfa5188684d893","side":"left"},{"sibling":"9e1128ce66c9847c130a04d2de905cc3f6bfb6519763c65cf85e8c1a2c4871a8","side":"left"},{"sibling":"9ebd81705c7903e17631922173cae6c0494ad086614d7cd5aff5313ac94a9893","side":"right"},{"sibling":"b31fd8919314f43e5528d32d692e1d0c9345f86173f69bd11bcf95d27af3147a","side":"right"},{"sibling":"762a50202e5bb846bfb5afec2f3cc80d9313546598c6c53f53acc51c6325b763","side":"left"},{"sibling":"e276c2896e885a069398e8350a2d9aae49ed0ab34771e352045341312283b40a","side":"right"},{"sibling":"b6709caadc8510310ee2ec0b66d1058fcad31c91c65cc2bad6f46a693d553580","side":"right"},{"sibling":"a72c3b8804a37d1a9d18e02e6fdb048bc8ea6b0746909cd2de10bdbabc793737","side":"left"},{"sibling":"803703dc2c50a646fa77b57c0056e9a5126611ba4bcde0d6013ccd6b2d44bdbf","side":"right"},{"sibling":"e78f244b1b8df6d5e3fdc6dd76b5c27d4e6fe3b93b8cd61355497b63d7e4cfe8","side":"left"},{"sibling":"6167cb552ed6871fbf0afcf3db01d1017af7d472b136fcbe5404d1df09f41cc1","side":"right"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":251380,"merkle_root":"e042805d07cd8dc777d49695ad78b8d4ec9721df271ff0245d7706773c30b4a5","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260626T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-26T05:37:59Z","sig_algorithm":"ed25519","signature":"c68a6e827804771acd244208495c5e35c6f417307db3c76192f038dedaa5b5f86e019074a3bea353357ff7457924c4f7907832638bb9a4d3d18e7a7f50c6a10b","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_da55e2a150ae77dd9c4cf13b13f44f065232322fc7e55d1890daafa5b006164a"}}