{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_44214d1a67260a0a7e66f584d796dd4c205ad11107e7ef3c56db4c9c34cf75a9","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_44214d1a67260a0a7e66f584d796dd4c205ad11107e7ef3c56db4c9c34cf75a9","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8dd7ce91fc27cbb6da32e5803f54d416178ca0c0851a2778fac3b7d46e87347f","published":"Mon, 18 May 2026 00:00:00 -0400","receipt_hash":"8dd7ce91fc27cbb6da32e5803f54d416178ca0c0851a2778fac3b7d46e87347f","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8dd7ce91fc27cbb6da32e5803f54d416178ca0c0851a2778fac3b7d46e87347f","observed_at":"2026-05-18T04:43:11.219741Z","parent_run_hash":"a8aad7414ebb6b75c726f09cd673410576a7f87e191fbdb9ddac99e9b2b95a05","published":"Mon, 18 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.15341v1 Announce Type: cross \nAbstract: LLMs are increasingly deployed in autonomous laboratories, under the assumption that their domain priors and reasoning over iterative feedback let them converge on good designs in fewer iterations than feedback-only baselines. Current iterative scientific design benchmarks, however, score only outcome snapshots at fixed horizons. This leaves the learning trajectory unmeasured, even though the trajectory is what captures learning efficiency, where each iteration saved is a real saving in cost and time. Motivated by this, we examine three evaluation choices that change the conclusions one draws about LLM learning efficiency in iterative scientific design: what to measure, what baseline to compare against, and what to ground against. We introduce LEAPBench, Learning Efficiency in Adaptive Processes, a 55-task framework that pairs a best-so-far area under the curve (AUC) trajectory metric with a classical Bayesian-optimization reference an","title":"LEAP: Trajectory-Level Evaluation of LLMs in Iterative Scientific Design","url":"https://arxiv.org/abs/2605.15341","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.15341v1 Announce Type: cross \nAbstract: LLMs are increasingly deployed in autonomous laboratories, under the assumption that their domain priors and reasoning over iterative feedback let them converge on good designs in fewer iterations than feedback-only baselines. Current iterative scientific design benchmarks, however, score only outcome snapshots at fixed horizons. This leaves the learning trajectory unmeasured, even though the trajectory is what captures learning efficiency, where each iteration saved is a real saving in cost and time. Motivated by this, we examine three evaluation choices that change the conclusions one draws about LLM learning efficiency in iterative scientific design: what to measure, what baseline to compare against, and what to ground against. We introduce LEAPBench, Learning Efficiency in Adaptive Processes, a 55-task framework that pairs a best-so-far area under the curve (AUC) trajectory metric with a classical Bayesian-optimization reference an","title":"LEAP: Trajectory-Level Evaluation of LLMs in Iterative Scientific Design","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-18T04:43:11Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.15341"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:8d82c087a00d07b1a7e22616c91e37f1e39ae303f50b0425224ee52e4bc90f6a722c481367cc623efca33329f90c94b32bfdea746e5631fd4559b918bc627105","signer":"crovia.substrate","subject":{"observed_at":"2026-05-18T04:43:11Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.15341"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"6bafa6e60716185e3601e2ba806d950a632ac090016e9d4d1809281614c9f03d","leaf_index":140578,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"709f22189b22d4857e1c1361e77259f1122598a8a5bb5c8e010f4aa94f1c1326","side":"right"},{"sibling":"8e3824c96149144efb8162eeb8de884a37a1b7e6ee81409387da48273cc294c2","side":"left"},{"sibling":"d63a9db67721f6544986393ee137ca3a23354f02c8043cf775b5d2e8ca7a3ee8","side":"right"},{"sibling":"269096a01d79345bfeb0d228e3b89b713fd01325c6303e4ac8d4cbcac9ec107c","side":"right"},{"sibling":"dead59a16a46ffb59968c18ba6918b1b54799c6d66ed856b1f99236d78ac0c57","side":"right"},{"sibling":"e927ee761b610b6646108660a1c36882541add1974b059be960ba3967a40c96b","side":"left"},{"sibling":"d10f43d638e7e4b5fe1190bff28b23af4dede5ac008b59c8c88d6e509f8c5960","side":"right"},{"sibling":"b29bcbd6adca3228d98aa21ca6aabf98e279340820cd7a960b41569223eebf29","side":"right"},{"sibling":"6bedf73520cf3dd8758d8bdedf3be245de9aea97abd42934aae25539176ae1b2","side":"left"},{"sibling":"07abc3bad689e74e6304772503dc9372a118e6f66883b8e88c43414efddac063","side":"right"},{"sibling":"28b78fb112bcf26b6801664db97eb8f52a9bccbf0a7ae6766e11845d443692df","side":"left"},{"sibling":"68d0a4634c1460a19c92edd9480df3aa733b814463e7420d1e14471bf61b2f83","side":"right"},{"sibling":"8af64f275b862349aa3bbb9d5cd7fa9a7fdd5620af3bf1b36b2a4519b0b53bdf","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"b98c2afadb358e5387e88f19588f8343a81b488d9b44a6f7e57a032db3a1b030","side":"right"},{"sibling":"11b0c1591747f09f7c8971a6caa19befcd81317ca9dfd417b143234df4e10c79","side":"right"},{"sibling":"87206f3bcc342797c990d87f7235c01f78d32ca59cfaf8ad18d71afc879ba477","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":140892,"merkle_root":"6cca56ead155990456b8a014cc50bddbe710f409b26e3d1bfa6fb12b0bfcf6bf","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260518T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-18T05:37:30Z","sig_algorithm":"ed25519","signature":"1e1135f7595f79b14fb11f5fa81a2e17ad31b11b44b427a5e40a7d511cd86447daf492babd368ab571cf26404c8c74c450d460130fca4b064eb2760367489a0f","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_44214d1a67260a0a7e66f584d796dd4c205ad11107e7ef3c56db4c9c34cf75a9"}}