{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d4c12cd4e7f1809b1abc9621f846908006fd68a1862052ba5468d81c8baba52f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d4c12cd4e7f1809b1abc9621f846908006fd68a1862052ba5468d81c8baba52f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"9f6d3cc23bff120326636da117d295d77bfe2942870cb8649fbd8fe16aa2186d","published":"Fri, 22 May 2026 00:00:00 -0400","receipt_hash":"9f6d3cc23bff120326636da117d295d77bfe2942870cb8649fbd8fe16aa2186d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"9f6d3cc23bff120326636da117d295d77bfe2942870cb8649fbd8fe16aa2186d","observed_at":"2026-05-22T04:43:12.593252Z","parent_run_hash":"dd4d56660d55b2d65dc84dc5e7c8f83487d90da2dd343b5707d6948a3bb0d917","published":"Fri, 22 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.20251v2 Announce Type: cross \nAbstract: Existing benchmarks for LLM coding agents primarily evaluate final outcomes. While useful for measuring overall capability, these metrics provide limited visibility and often miss defects that arise during execution. We present ProcBench, a benchmark for execution-process evaluation in LLM coding agents. ProcBench organizes recurrent execution defects into a reusable ontology covering 11 defect types in 4 categories, and evaluates agent trajectories through standardized process evidence rather than final outcomes alone. To support comparison across heterogeneous agents, ProcBench standardizes raw logs into a unified trajectory representation and reports calibrated scorecards over process-level findings. In addition, ProcBench uses control preservation as a way to quantify execution-process quality, capturing whether execution remains interpretable, interruptible, correctable, reversible, and able to hand back authority when needed. We ","title":"ProcBench: Evaluating Process-Level Defects and Control Preservation in LLM Coding Agents","url":"https://arxiv.org/abs/2605.20251","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.20251v2 Announce Type: cross \nAbstract: Existing benchmarks for LLM coding agents primarily evaluate final outcomes. While useful for measuring overall capability, these metrics provide limited visibility and often miss defects that arise during execution. We present ProcBench, a benchmark for execution-process evaluation in LLM coding agents. ProcBench organizes recurrent execution defects into a reusable ontology covering 11 defect types in 4 categories, and evaluates agent trajectories through standardized process evidence rather than final outcomes alone. To support comparison across heterogeneous agents, ProcBench standardizes raw logs into a unified trajectory representation and reports calibrated scorecards over process-level findings. In addition, ProcBench uses control preservation as a way to quantify execution-process quality, capturing whether execution remains interpretable, interruptible, correctable, reversible, and able to hand back authority when needed. We ","title":"ProcBench: Evaluating Process-Level Defects and Control Preservation in LLM Coding Agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-22T04:43:12Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.20251"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:af589fc98ee652c4b0614ef7247b8820bf02f0ea3a5459041d173e5b191c64ce16cb07f68d1dbf8206e9e3315e1f39666300846ee6eff4d547d3603b661cf001","signer":"crovia.substrate","subject":{"observed_at":"2026-05-22T04:43:12Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.20251"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d266d62f95e820dcf295671b93f8c68efa8757be8e985a803fcc3827808a0dc4","leaf_index":148177,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"acfb5298640d555ab595a7fa26e8ac61247a71f558e749b937f4b033cab8a60b","side":"left"},{"sibling":"0283fa6cddeaf9adf32209c028567194f6394cf41f8cce85138af4f6751be27d","side":"right"},{"sibling":"700bfb80ac78ad01d4e4895b72f282e48b67120f554b028c393d78bf6386eab6","side":"right"},{"sibling":"77466f444487fbe1510677b2c2d99cb388d4594ccf5ecd926031232d828bd4c9","side":"right"},{"sibling":"e515ecd738733f891313e29360912c2fd7b26e7dda403946d6d8258b563289fd","side":"left"},{"sibling":"0fdd2a5a527954fdd3c361b2337a8a7abc44029bc78402b0f1c47428a708f845","side":"right"},{"sibling":"9041a5a271687216ad8ad42ade0fd8c711a35ed68eb350b81aed4384188eec3c","side":"left"},{"sibling":"ce8423e7b33fd98ad2188ec515d860afebe3ce0e74717c41376c90ea6acfb384","side":"left"},{"sibling":"c5d582bc1cdd6d6494fd9e29c8b4aadd02d777f7ef92fc6d4afad7ee42785e79","side":"right"},{"sibling":"d337a9fdcfc121e9691d8db9173af6a3fe0c33d4a6d5f0c8a7a01af04f9fb856","side":"left"},{"sibling":"8b39e07457f5cc5d687d2ae42284dbe705bb87084e7db4b626aff81e51dacd19","side":"right"},{"sibling":"79a713e1e345ccb99c5fe994a11708c8e9bcfa2e91f940d70421cb7d8d77ecc6","side":"right"},{"sibling":"249870fb494bef050c409081e5de45f9042938d7b2823524ee496f296aa63667","side":"right"},{"sibling":"96c48ee8328f1b7925a4cc4421df5cb0bd81c92a5d8354c93126fa5f0166d225","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"4e13b4a3e69bb83d13913477a782913ce03937edd65046853c1964d3cbb6564b","side":"right"},{"sibling":"0f7b2df1c4580bf7bb7c24b9158ba20593a06af18a5f18d0973e5eff20c35cd8","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":148601,"merkle_root":"44900cffd986f40535c83f46a250e86fbf1019d41f27080a00fbf9b8d77ec33a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260524T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-24T13:37:32Z","sig_algorithm":"ed25519","signature":"059e428c3c5241de303721ad6ac7b748758372180f3f0a717810312aecd6fab073abb3ea264157666de581a2b361c4c6e2aa8ca081b08ce9be090721f0e3400e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d4c12cd4e7f1809b1abc9621f846908006fd68a1862052ba5468d81c8baba52f"}}