{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_e680b3f1ab292148c76dd6ca6429f789cd89aa2c0aca7570e40802d09500c91d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_e680b3f1ab292148c76dd6ca6429f789cd89aa2c0aca7570e40802d09500c91d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f98a075120f29f5f4e4333da37bc77b860447806c4fb9172f302f08dcb79acdf","published":"Fri, 15 May 2026 00:00:00 -0400","receipt_hash":"f98a075120f29f5f4e4333da37bc77b860447806c4fb9172f302f08dcb79acdf","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f98a075120f29f5f4e4333da37bc77b860447806c4fb9172f302f08dcb79acdf","observed_at":"2026-05-15T04:43:17.611638Z","parent_run_hash":"5c64f85625fabd323e9c4a1cf068c012fb88a248deda9a9ac702fb2f9799f2e5","published":"Fri, 15 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.13950v1 Announce Type: cross \nAbstract: Autonomous language-model agents are increasingly evaluated on long-horizon tool-use tasks, but existing benchmarks rarely capture the complexity and nuance of real scientific work. To address this gap, we introduce Collider-Bench, a benchmark for evaluating whether LLM agents can reproduce experimental analyses from the Large Hadron Collider (LHC) using only public papers and open scientific software. Such analyses are often difficult to reproduce because the public toolchain only approximates the software used internally by the experimental collaborations, while the published papers inevitably omit implementation details needed for a faithful reconstruction. Agents must therefore rely on physical reasoning, domain knowledge, and trial-and-error to fill these gaps. Each task requires the agent to turn a published analysis into an executable simulation-and-selection pipeline and submit predicted collision event yields in specified sign","title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","url":"https://arxiv.org/abs/2605.13950","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.13950v1 Announce Type: cross \nAbstract: Autonomous language-model agents are increasingly evaluated on long-horizon tool-use tasks, but existing benchmarks rarely capture the complexity and nuance of real scientific work. To address this gap, we introduce Collider-Bench, a benchmark for evaluating whether LLM agents can reproduce experimental analyses from the Large Hadron Collider (LHC) using only public papers and open scientific software. Such analyses are often difficult to reproduce because the public toolchain only approximates the software used internally by the experimental collaborations, while the published papers inevitably omit implementation details needed for a faithful reconstruction. Agents must therefore rely on physical reasoning, domain knowledge, and trial-and-error to fill these gaps. Each task requires the agent to turn a published analysis into an executable simulation-and-selection pipeline and submit predicted collision event yields in specified sign","title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-15T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.13950"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ebf097f30aa1ac87d28668ceeefcdd5d7bcfbf0f09be5a273f7362ddff702b5a657809cdd9b2632c4b96f5c5bad7c0fe7b268fc8813e41e4fd5d8df321ae960a","signer":"crovia.substrate","subject":{"observed_at":"2026-05-15T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.13950"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"29c9aa4bd287db0b0e3aa3f7d948969c24110e0163ff2e670b13d46fd61aa315","leaf_index":134517,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"8c7f7a5abaddd75d4b7d7fc13f458cd9ca2524c817e6e8c4fb3e477c3573db8d","side":"left"},{"sibling":"3f5fab259ae8aa8cbd587985e25701fa5b0eb0bb9c342f31bc2c22a41f13194f","side":"right"},{"sibling":"9106933d8e309d19b69b99b81245161398d53fe9a5f3a8867c14783ee71d6981","side":"left"},{"sibling":"b35bfab17c0e83e9c4a8946369d316336e8662368c9aed2b05ce86081560ba85","side":"right"},{"sibling":"fff87ab95770239012633ad0c4c704d70b78a2607b5f0ecd6e39ae89ddda365f","side":"left"},{"sibling":"621b9f34daa668dc5e3535facb5a01cc0190c3cd1989b306a638e48259c4a25a","side":"left"},{"sibling":"3262b224bf97e5bf6c5eed220ac5a0ff7a7c4e3b809039f633c552438e392b8c","side":"left"},{"sibling":"74f15de1b6fc788befd85007692dbe91c695ee2181b0b20f20e5a3e4147eaa67","side":"right"},{"sibling":"4711a7f4329f1874c3aa1ae93c336e1d4fa402bdd4b3d766c2a95304ea226882","side":"left"},{"sibling":"8ce2a4687a8ceb409df2e1cb10e44a610b21dc294e518a8550b6d1122335ca2b","side":"right"},{"sibling":"36672459e5ed50c64ee1842b69cb6d2eb682c2a04844555be8d124257571994a","side":"left"},{"sibling":"727783827652adfa99c455bd80a01bfb33836228e51068b4f654ef3da468ca69","side":"left"},{"sibling":"623194cd30880ed223e306737fdb111aa0d781751bfc47553c404a6af6aad2c4","side":"right"},{"sibling":"fc8f53ed42756907fb79ee19a4ed09f72c560e5302b3d98198b96bf1da635a4a","side":"right"},{"sibling":"d6607539da7ba39ec68be2d12f27ed6768766c745e3120fd915f88c5e288e07c","side":"right"},{"sibling":"b63408a424d27cd6a75e0fb155e69a58a328f41e9cb9dba1eddef9a5289cc7fd","side":"right"},{"sibling":"356fb36a4e188f03d7a05c54cd8789bdd40eda454b9bc9560f667acc08e6c4e0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":134886,"merkle_root":"c6c7ae28c065bced89e7f844216b073f1a7cc4b378db0d41a98bcd21b28066db","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260515T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-15T05:37:26Z","sig_algorithm":"ed25519","signature":"5a3978c26017daf4104adbb3e1c3099c5750157acbfc7242ece1815dc6740fe08a690291ce0afe42011e20cc565b5ebe64ec016bf658b5bab563a63337985c05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_e680b3f1ab292148c76dd6ca6429f789cd89aa2c0aca7570e40802d09500c91d"}}