{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_938f90e5e0bbc19794a309bfb669f85706f22f27ee3b7dae67546fbe9ba296ea","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_938f90e5e0bbc19794a309bfb669f85706f22f27ee3b7dae67546fbe9ba296ea","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"61053f1aa7d1114537fd128884a581723ae51f9937e5e565320ccaa66fb37dee","published":"Tue, 14 Jul 2026 00:00:00 -0400","receipt_hash":"61053f1aa7d1114537fd128884a581723ae51f9937e5e565320ccaa66fb37dee","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"61053f1aa7d1114537fd128884a581723ae51f9937e5e565320ccaa66fb37dee","observed_at":"2026-07-14T04:43:37.834979Z","parent_run_hash":"66b89520a448b8d9fe7d8f602ef38b82b6c95e57f532ce72de51375b41870477","published":"Tue, 14 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.02905v2 Announce Type: replace \nAbstract: Autonomous agents powered by large language models (LLMs) promise to accelerate scientific discovery end-to-end, but rigorously evaluating their capacity for verifiable discovery remains a central challenge. Existing benchmarks face a trade-off: they either heavily rely on LLM-as-judge evaluations of automatically generated research outputs or optimize convenient yet isolated performance metrics that provide coarse proxies for scientific insight. To address this gap, we introduce FIRE-Bench (Full-cycle Insight Rediscovery Evaluation), a benchmark that evaluates agents through the rediscovery of established findings from recent, high-impact machine learning research. Agents are given only a high-level research question extracted from a published, verified study and must autonomously explore ideas, design experiments, implement code, execute their plans, and derive conclusions supported by empirical evidence. We evaluate a range of sta","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","url":"https://arxiv.org/abs/2602.02905","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.02905v2 Announce Type: replace \nAbstract: Autonomous agents powered by large language models (LLMs) promise to accelerate scientific discovery end-to-end, but rigorously evaluating their capacity for verifiable discovery remains a central challenge. Existing benchmarks face a trade-off: they either heavily rely on LLM-as-judge evaluations of automatically generated research outputs or optimize convenient yet isolated performance metrics that provide coarse proxies for scientific insight. To address this gap, we introduce FIRE-Bench (Full-cycle Insight Rediscovery Evaluation), a benchmark that evaluates agents through the rediscovery of established findings from recent, high-impact machine learning research. Agents are given only a high-level research question extracted from a published, verified study and must autonomously explore ideas, design experiments, implement code, execute their plans, and derive conclusions supported by empirical evidence. We evaluate a range of sta","title":"FIRE-Bench: Evaluating AI Agents on the Rediscovery of Scientific Insights","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-14T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.02905"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:d0c7cd768cffbb98626b07e5e52fbf05f7fe8e3c50a0cb676187e797e6118454100d3aeced5bc389a38bf098fe8a1a2e9640b5f3ed53bc58728ed69334cb8b04","signer":"crovia.substrate","subject":{"observed_at":"2026-07-14T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.02905"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"a04cf643797520f8aa75f6d68aeced2d302295a3c8fe9168ec7cd8ae231c556c","leaf_index":313122,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"d817a2fb3b80a581b7a8cd046f43d4ea33448b6b1e77461ff24fffc6af2a3a4f","side":"right"},{"sibling":"68da250974f327818b2c8c89491df7f9dcbbdea5cadc981c29ca3cd582768a3e","side":"left"},{"sibling":"9e83ab38b9b21658e5318c2e59c6a4654814b9f1d9b52c225ee79b3b1c98eba0","side":"right"},{"sibling":"37c07ee1f458143d4786ade9ddda58a4f458f55ca9e384bdae8bf804fd1a37e6","side":"right"},{"sibling":"cc8bf68ae0277149de11f9418c8ca27270cb49121a0e31f2b4c553e855efdfa9","side":"right"},{"sibling":"63c17f44c675a07d319f8cc451137e3f1ec8aa5e4b9afb9f94042cbadea1a6d3","side":"left"},{"sibling":"a833523d950428dc21ca9a8e4dda7f94685e705281058be8f40306936c6e2602","side":"right"},{"sibling":"edf133a890c50dc04a7d212f979f1424434d765ec9a191894fda402b340d9450","side":"right"},{"sibling":"ae8fbfdec46ce02d0a92359c6a034f0439394d13e731da90290f5ca7192e6da4","side":"left"},{"sibling":"8393dc03644000353c5d503b6ec261c022cd2aa85a649223916153bc8af2dc75","side":"left"},{"sibling":"1627ce6908965f2e5fbc7c16bc8e247867478c587014561d0de1359dc2dc0afc","side":"left"},{"sibling":"74e1ba9fd48bdd0f52777dd5b56c10706e73f42629013748eca13ef113cd59db","side":"right"},{"sibling":"682fd39539db5bce908d0ab6f180374728fed6b34a9a8c87c7b4eb5a726e30b8","side":"right"},{"sibling":"184927f71d667ac0c6ebeadb33394626973738b9107c6dc3c0c4949b44acf295","side":"right"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"c4c189d79669979b59d9aa976ee249c54a7e065b4a05bf49463459cc7f37428e","side":"right"},{"sibling":"19475e206bdf2698769a286db2c97c4d3f089741319f9b29a139038c6e511975","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":313381,"merkle_root":"5f7d48580373d0ab0bc6bdf34f26de442f9a86d130946f7ee45addd1a55cb9f4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260714T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-14T05:38:23Z","sig_algorithm":"ed25519","signature":"97364224142ed1b2a8925fed1bfba2e0a8a6a15f8ca0999b934846ad83584a78c7f0f5cd09dbc1c938b35383fcc5c8fa4b723118e30916628879d1c5fd3a9908","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_938f90e5e0bbc19794a309bfb669f85706f22f27ee3b7dae67546fbe9ba296ea"}}