{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_bb61547fb8b1aaa475243d63c0f2d52a1bb91f9e3cb77dbd08eb8a47a72feb65","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_bb61547fb8b1aaa475243d63c0f2d52a1bb91f9e3cb77dbd08eb8a47a72feb65","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"820ac2b3a705c2141bfdbfe9f6172229ef9d5bc6ad89dc32a019f0d5d7e49e53","published":"Fri, 17 Jul 2026 00:00:00 -0400","receipt_hash":"820ac2b3a705c2141bfdbfe9f6172229ef9d5bc6ad89dc32a019f0d5d7e49e53","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"820ac2b3a705c2141bfdbfe9f6172229ef9d5bc6ad89dc32a019f0d5d7e49e53","observed_at":"2026-07-17T04:43:38.280949Z","parent_run_hash":"113193614a8af99887180226d4e28a8b71d957da5fe3694f0e7a56807c145504","published":"Fri, 17 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.17930v3 Announce Type: replace \nAbstract: AI evaluations are shifting toward harder tasks that benefit from longer trajectories involving tool use and iterative problem solving. As a result, performance is increasingly sensitive to the amount and allocation of compute available at test time (\"inference compute\"). Yet many evaluations still report performance at a single restrictive budget, meaning that low scores may reflect the evaluation setup rather than the model's underlying capability. To test this, we evaluate up to 12 frontier language models on seven challenging benchmarks spanning software engineering, mathematics, medicine, and cybersecurity. We use a controlled setup combining three simple inference-scaling interventions: larger token budgets, context compaction, and repeated submission attempts, guided either by the model itself or by minimal correctness feedback. We find three main results. First, larger token budgets substantially improve performance on benchm","title":"How Inference Compute Shapes Frontier LLM Evaluation","url":"https://arxiv.org/abs/2606.17930","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.17930v3 Announce Type: replace \nAbstract: AI evaluations are shifting toward harder tasks that benefit from longer trajectories involving tool use and iterative problem solving. As a result, performance is increasingly sensitive to the amount and allocation of compute available at test time (\"inference compute\"). Yet many evaluations still report performance at a single restrictive budget, meaning that low scores may reflect the evaluation setup rather than the model's underlying capability. To test this, we evaluate up to 12 frontier language models on seven challenging benchmarks spanning software engineering, mathematics, medicine, and cybersecurity. We use a controlled setup combining three simple inference-scaling interventions: larger token budgets, context compaction, and repeated submission attempts, guided either by the model itself or by minimal correctness feedback. We find three main results. First, larger token budgets substantially improve performance on benchm","title":"How Inference Compute Shapes Frontier LLM Evaluation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-17T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.17930"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:d3792931ca8dc6342602bf9eb1a33b667600dae7eed01d7c8cff24366debb1147ea569af2d5d6020b6e7b019dbc048c905fc9daed4ab8e4ff4055f46d8a8a902","signer":"crovia.substrate","subject":{"observed_at":"2026-07-17T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.17930"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"4859c79b43f9a69f64162ede078954833b6763ba451f899915566fd5dd62a110","leaf_index":323190,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"3ca0677547bba65a1e1389aeff499f27cdb41a09ce00e5f58e505e7a1b6b1112","side":"right"},{"sibling":"1b4f7268632e78ccb816091dd4bcadcd0ad78012a00e89159e36ec348cc4cd01","side":"left"},{"sibling":"940ef5b0b7476ed6f99e1170e85a67665b1c25088b9542f8cb440244092cfacc","side":"left"},{"sibling":"80f5bad79209cef3443440037d8c51fa5f9a584f294fb7781bb5635489926500","side":"right"},{"sibling":"43762f5297f1a9ad4b405571d2739158e92f0d62faec0cb5884f84fb597bedf1","side":"left"},{"sibling":"00bd52f3534c19833a0f638a09d0c0670180e59cb7ba9a2aeb3b4c000ac66696","side":"left"},{"sibling":"65f8a681723598b39869e2bf9640e4e77ef6eea44f85f6f14b4e77bf40cd5f40","side":"left"},{"sibling":"de6ef12c0ebc7e3e48430c64265163103eb92448efd698dbc2b104ada5a5389d","side":"right"},{"sibling":"4242cc570ec8c36a37f3f6f20dcae20b49fccc17c7b53ba47d8715e465bec585","side":"right"},{"sibling":"be025f48721bfc0ca7107f0454bda3ab460e50539f0caeb1bf839a8dabcf036c","side":"left"},{"sibling":"05a09763743cdc09fc45cf454e4e3ea4a0d1cd74f9c8162b2a57e2c873160908","side":"left"},{"sibling":"de3120ef2488b8a791a686b47257da4e612256abdfcdda7519265e7edd47d041","side":"left"},{"sibling":"e86f56a4883492da5b5e7b0201324c52946e865e69b99ebb532f41fe3c658ee4","side":"right"},{"sibling":"34d85f6ad6cc7dfa79d90e2b9ff99a561bcdc75b0301bbbd3e83861f54535c1e","side":"left"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"9b11714124b9b951ff9450b0ee9d625a0b70b6da2cf388bd3df1475eec0b17ba","side":"right"},{"sibling":"a4523a9014d45df43e006e9210a73428c380d771f2c650a1b986910759b0cdf7","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":323382,"merkle_root":"2f4d32419c80a9600aba5a480fc3fb7012ec0a695c91a1b055048e78760b65ca","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260717T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-17T05:38:31Z","sig_algorithm":"ed25519","signature":"495308c7bf004117807331d3f71d0b079f6bd7ed7737faa58c773b1b3e80ee84928d2d8519cc7d2b809506501aada1be6546f72c7ecda3dad5445cbb44502209","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_bb61547fb8b1aaa475243d63c0f2d52a1bb91f9e3cb77dbd08eb8a47a72feb65"}}