{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_08bf20f7c1cd3a3946020d871840f51153befe515dcf80fbc9a298467304e54d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_08bf20f7c1cd3a3946020d871840f51153befe515dcf80fbc9a298467304e54d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f8da934c3229be47d8079810c6585e3682c18d4380cb9bb6b1bcd9335fdf4789","published":"Fri, 03 Jul 2026 00:00:00 -0400","receipt_hash":"f8da934c3229be47d8079810c6585e3682c18d4380cb9bb6b1bcd9335fdf4789","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f8da934c3229be47d8079810c6585e3682c18d4380cb9bb6b1bcd9335fdf4789","observed_at":"2026-07-03T04:43:38.241623Z","parent_run_hash":"f0e30469786257a5e74170498cacb4c028623bf32d6d06d4dbadc488960545be","published":"Fri, 03 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.02032v1 Announce Type: new \nAbstract: Evaluating LLM agents on benchmarks like SWE-Bench and GAIA can be expensive, time-consuming, and requires complex infrastructure. A single evaluation can cost thousands of dollars and take days to complete. In contrast, non-agentic LLM benchmarks that test individual capabilities (e.g., reasoning, code generation) are fast and cheap to run. In this paper, we investigate whether performance on expensive agentic benchmarks can be accurately predicted by the performance on a small, carefully selected subset of atomic evaluation instances. We introduce PACE, a framework that constructs proxy benchmarks by selecting instances from existing non-agentic evaluations whose aggregate scores most reliably predict model performances on agentic benchmarks. Given a pool of candidate instances spanning atomic capabilities, PACE fits a regression that maps a model's scores on a compact subset of source instances to its score on the target agentic bench","title":"PACE: A Proxy for Agentic Capability Evaluation","url":"https://arxiv.org/abs/2607.02032","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.02032v1 Announce Type: new \nAbstract: Evaluating LLM agents on benchmarks like SWE-Bench and GAIA can be expensive, time-consuming, and requires complex infrastructure. A single evaluation can cost thousands of dollars and take days to complete. In contrast, non-agentic LLM benchmarks that test individual capabilities (e.g., reasoning, code generation) are fast and cheap to run. In this paper, we investigate whether performance on expensive agentic benchmarks can be accurately predicted by the performance on a small, carefully selected subset of atomic evaluation instances. We introduce PACE, a framework that constructs proxy benchmarks by selecting instances from existing non-agentic evaluations whose aggregate scores most reliably predict model performances on agentic benchmarks. Given a pool of candidate instances spanning atomic capabilities, PACE fits a regression that maps a model's scores on a compact subset of source instances to its score on the target agentic bench","title":"PACE: A Proxy for Agentic Capability Evaluation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-03T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.02032"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4962a31dd766246aafbf9773748f46ab43e479fe87bc256f146464500010646015514f4a5dac1c03b99e277b81867e931602650437574c9dcdbb95f4bf2cd000","signer":"crovia.substrate","subject":{"observed_at":"2026-07-03T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.02032"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"540ffa73dda59d692a41bcae998aaf2e6495ef177dff7c25c4333e89086bbd71","leaf_index":275393,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"98f51e626b86c9ba6c0b9b71d0b6a8f94776f323d612d84083bdef1d4965604e","side":"left"},{"sibling":"bcc1b35cf3d9e43a3b49d9ee631b74d4d31424317bbb3dca81b6e804e519b13b","side":"right"},{"sibling":"abba8d450331da29936821bc1184a0f4ad74c2143aabe79b848bb135c66cb44a","side":"right"},{"sibling":"1da9d236c72cc5e1121662636d9c0a7ce46d55227287a77f4b9440d887af08dc","side":"right"},{"sibling":"5d8a5029c968d0f123b5bd1701855306d2cd60dd310556cfb484e937c3d1cde5","side":"right"},{"sibling":"849f8f2e2bcf09437f8fb74995be42a57cbd6b5bbf3b4d1b680700e05d9f3916","side":"right"},{"sibling":"3a4dd616b0a708226bdfc1375ba2f51d5b9f5745b3184702e9cbca38e27d2a27","side":"left"},{"sibling":"fd055e725b36f5bebbaeb18583f9d7011c4d8c2a608a11cbd2c166cd895e7a85","side":"left"},{"sibling":"e1a65581e9d68211aa8ba0d138c8d2a4e80afd551e6d1b00c4575c39085df705","side":"left"},{"sibling":"92c062f377cd53076b8dfe55c25ab217059e20113433673cc5747b9348b07e01","side":"left"},{"sibling":"e36f7633c67452f41a7377a7bec9b2d454442688d26539321eebf92cae9db0f0","side":"right"},{"sibling":"4dbd8247ba08a5432c7d6540711da9acb2f59e6189865aa8552dee37f69286a9","side":"right"},{"sibling":"41cd1885dc3fcb51e49eeb887d6d22ec2cfa58df0e4f8d7c7dddf3a1b0ce8249","side":"left"},{"sibling":"8a09562f6b247c1c3cd1fea36cb3b8f1cf5c575479dd514573856a380a964bf5","side":"left"},{"sibling":"723981908169653ca6d835aa9b8381a8c7ad3e3e3830d0792bc32032cda615ee","side":"right"},{"sibling":"c0594fa1ee81d5f019cccc7b5e51af603c6d7e43995498c451012060c7d06165","side":"right"},{"sibling":"4de6a2fb22efbb50c84dc62abeb0f2cbc8c663a9540aeba9e758ebfdfe3e86dd","side":"right"},{"sibling":"fdbb3519f8dc411a4043dfb5abdbfea5441e130326183ac2247c42584033f152","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":275799,"merkle_root":"2581d0d6e5fa345cdf2e8ab3b191ace76d6b14189901ab0e4c2291ca1d1ae1e6","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260703T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-03T05:38:09Z","sig_algorithm":"ed25519","signature":"e44a386a5ae00c0e7fc67b1179bb9060bf0fefc006e454fec70e27668182ff497d1e2faf0c3b22de0917beefb7c80e880dae925a3f67e1b16aa0eb44bf947407","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_08bf20f7c1cd3a3946020d871840f51153befe515dcf80fbc9a298467304e54d"}}