{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_fe4776c24053259afe96cc5cb61d6da3a6a350ce954b0466294bd1906fdc2848","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_fe4776c24053259afe96cc5cb61d6da3a6a350ce954b0466294bd1906fdc2848","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"66168260a8220eb1ba2514b8e12d89d7ec034ff1cd58b2902d4e25fca4255006","published":"Fri, 15 May 2026 00:00:00 -0400","receipt_hash":"66168260a8220eb1ba2514b8e12d89d7ec034ff1cd58b2902d4e25fca4255006","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"66168260a8220eb1ba2514b8e12d89d7ec034ff1cd58b2902d4e25fca4255006","observed_at":"2026-05-15T04:43:17.611638Z","parent_run_hash":"5c64f85625fabd323e9c4a1cf068c012fb88a248deda9a9ac702fb2f9799f2e5","published":"Fri, 15 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.14133v1 Announce Type: new \nAbstract: Interactive agent benchmarks face a tension between scalable construction and realistic workflow evaluation. Hand-authored tasks are expensive to extend and revise, while static prompt evaluation misses failures that only appear when agents operate over persistent state. Existing interactive benchmarks have advanced agent evaluation significantly, but most initialize tasks from clean state and do not systematically test how agents handle pre-existing partial, stale, or conflicting artifacts. We present \\textbf{ClawForge}, a generator-backed benchmark framework for executable command-line workflows under state conflict. The framework compiles scenario templates, grounded slots, initialized state, reference trajectories, and validators into reproducible task specifications, and evaluates agents step by step over persistent workflow surfaces using normalized end state and observable side effects rather than exact trajectory matching. We ins","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","url":"https://arxiv.org/abs/2605.14133","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.14133v1 Announce Type: new \nAbstract: Interactive agent benchmarks face a tension between scalable construction and realistic workflow evaluation. Hand-authored tasks are expensive to extend and revise, while static prompt evaluation misses failures that only appear when agents operate over persistent state. Existing interactive benchmarks have advanced agent evaluation significantly, but most initialize tasks from clean state and do not systematically test how agents handle pre-existing partial, stale, or conflicting artifacts. We present \\textbf{ClawForge}, a generator-backed benchmark framework for executable command-line workflows under state conflict. The framework compiles scenario templates, grounded slots, initialized state, reference trajectories, and validators into reproducible task specifications, and evaluates agents step by step over persistent workflow surfaces using normalized end state and observable side effects rather than exact trajectory matching. We ins","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-15T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.14133"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:bed9333f5167dd6c3f1097d0ecf951baf7186539bf9fe9c8dad90423070db92c59d821af091d903a5b0a7795f4b67161dd7dde6bfcbb81b515d1035fb11e9c01","signer":"crovia.substrate","subject":{"observed_at":"2026-05-15T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.14133"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"eaf8cf4b2126db015a31be1eabfd3f72157357f55765975a8eb081f160d91f5e","leaf_index":134413,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"9ff74bd3f1d002c6f2322898d82b00555eaf223207eb61e9d059164220c8d4d4","side":"left"},{"sibling":"e662dff347445ecb1c942959864bc19369777ab0f56225e1aeaceed036ac2bd8","side":"right"},{"sibling":"c0dd43fe25da757a38efb27b3e46d969bb33cc84e58f3051cdb308d667b19682","side":"left"},{"sibling":"c7cc5c6c41fa51d13fa15d24ff5056173b0fae557a82caf6ea95ce2937ee6d6f","side":"left"},{"sibling":"13d8c108165fa0a59894ebffdbcf8d19272a28593e4f40fe265d42e41e74e147","side":"right"},{"sibling":"47e3ac984506f746cb9b8fce8892d4772feb17a4ade507717189b67cee9f89f2","side":"right"},{"sibling":"95db707d4c2fb29f0e7091946b3cabe291908aa524bcc83e082b0ed9c09c741c","side":"right"},{"sibling":"74f15de1b6fc788befd85007692dbe91c695ee2181b0b20f20e5a3e4147eaa67","side":"right"},{"sibling":"4711a7f4329f1874c3aa1ae93c336e1d4fa402bdd4b3d766c2a95304ea226882","side":"left"},{"sibling":"8ce2a4687a8ceb409df2e1cb10e44a610b21dc294e518a8550b6d1122335ca2b","side":"right"},{"sibling":"36672459e5ed50c64ee1842b69cb6d2eb682c2a04844555be8d124257571994a","side":"left"},{"sibling":"727783827652adfa99c455bd80a01bfb33836228e51068b4f654ef3da468ca69","side":"left"},{"sibling":"623194cd30880ed223e306737fdb111aa0d781751bfc47553c404a6af6aad2c4","side":"right"},{"sibling":"fc8f53ed42756907fb79ee19a4ed09f72c560e5302b3d98198b96bf1da635a4a","side":"right"},{"sibling":"d6607539da7ba39ec68be2d12f27ed6768766c745e3120fd915f88c5e288e07c","side":"right"},{"sibling":"b63408a424d27cd6a75e0fb155e69a58a328f41e9cb9dba1eddef9a5289cc7fd","side":"right"},{"sibling":"356fb36a4e188f03d7a05c54cd8789bdd40eda454b9bc9560f667acc08e6c4e0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":134886,"merkle_root":"c6c7ae28c065bced89e7f844216b073f1a7cc4b378db0d41a98bcd21b28066db","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260515T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-15T05:37:26Z","sig_algorithm":"ed25519","signature":"5a3978c26017daf4104adbb3e1c3099c5750157acbfc7242ece1815dc6740fe08a690291ce0afe42011e20cc565b5ebe64ec016bf658b5bab563a63337985c05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_fe4776c24053259afe96cc5cb61d6da3a6a350ce954b0466294bd1906fdc2848"}}