{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_3d1a7ad2700eee5b3a0607ffb9b314bf23f3d3897e9a2c58dc4e40deffd945cd","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_3d1a7ad2700eee5b3a0607ffb9b314bf23f3d3897e9a2c58dc4e40deffd945cd","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6c85441f65d022ba062fa5ffb8d0086d420652ea6ba8a57b8380e5c017d0cc95","published":"Wed, 03 Jun 2026 00:00:00 -0400","receipt_hash":"6c85441f65d022ba062fa5ffb8d0086d420652ea6ba8a57b8380e5c017d0cc95","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6c85441f65d022ba062fa5ffb8d0086d420652ea6ba8a57b8380e5c017d0cc95","observed_at":"2026-06-03T04:43:57.136784Z","parent_run_hash":"62ae9c8eda846b00bc49666345b338d00756c5203438666c4c1fc694cc364b84","published":"Wed, 03 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.02060v2 Announce Type: replace \nAbstract: Deep-research agents solve tasks through long trajectories of search, tool use, evidence inspection, and answer synthesis. Evaluation based on final answers shows whether an agent succeeds, but not which parts of the trajectory make the answer unreliable. We study span-level error localization for deep-research agents. We collect 2,790 real trajectories from two agent frameworks, three backbone models, and three benchmarks, convert raw logs into semantic spans, and annotate harmful error spans through LLM-assisted expert review. From these annotations, we build TELBench, a 1,000-instance benchmark for identifying error spans among normal exploration, failed searches, tentative hypotheses, and harmless noise. We further propose DRIFT, a claim-centric auditing framework that tracks agent claims, checks their support in trajectory evidence, and marks spans where unsupported or conflicting claims affect the answer path. Experiments acros","title":"Where Do Deep-Research Agents Go Wrong? Span-Level Error Localization in Agent Trajectories","url":"https://arxiv.org/abs/2606.02060","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.02060v2 Announce Type: replace \nAbstract: Deep-research agents solve tasks through long trajectories of search, tool use, evidence inspection, and answer synthesis. Evaluation based on final answers shows whether an agent succeeds, but not which parts of the trajectory make the answer unreliable. We study span-level error localization for deep-research agents. We collect 2,790 real trajectories from two agent frameworks, three backbone models, and three benchmarks, convert raw logs into semantic spans, and annotate harmful error spans through LLM-assisted expert review. From these annotations, we build TELBench, a 1,000-instance benchmark for identifying error spans among normal exploration, failed searches, tentative hypotheses, and harmless noise. We further propose DRIFT, a claim-centric auditing framework that tracks agent claims, checks their support in trajectory evidence, and marks spans where unsupported or conflicting claims affect the answer path. Experiments acros","title":"Where Do Deep-Research Agents Go Wrong? Span-Level Error Localization in Agent Trajectories","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-03T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.02060"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:44493ed2bbcdf2ab668eef3c061747edc9a536391ed44f2fb777ac4edfdaf8f8854168c62bfd7ca3250e51852899a9e48a96e78781e4bfdcd93f6f873a3e6f03","signer":"crovia.substrate","subject":{"observed_at":"2026-06-03T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.02060"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"fc640400b3761708091c23bbcb230c881ee1d99c9da39aef52e535ad18cc0603","leaf_index":209339,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"70879ec0663a4c157a3c569e6e738e550e2a1804e59af901152efe9b40085bbd","side":"left"},{"sibling":"fb31a8558a0e63dd109c4b1d574ba7a5dde13369292e48e861679830a119c31e","side":"left"},{"sibling":"7e579ac0ddea672dc7e82e98a40fa4a5f4952c2fa8701ef9d45159789065afac","side":"right"},{"sibling":"29be323356642caccde1dfa2c64d34861e55c08842d71926bb9eb3d563bedab1","side":"left"},{"sibling":"d1d3330fb955b3c8d5fa71bad6b0df3cdd41176cce6f0da94f83827e118645f4","side":"left"},{"sibling":"0ee55be4f19504f49b1520a00b7c41813b27148ebdf7c2099670dc547019fccb","side":"left"},{"sibling":"4a9041cff48b223827c44d2d56d150c595450d4a6390b4eb331519bc36b634f1","side":"right"},{"sibling":"768ffa75e0a020ca1e4038beeef0da5c26e64a7ba3a0528188d49b4d35664067","side":"left"},{"sibling":"2b9867040ec52d22722edc26b204e15651723dcc66d361cc1af11490a854d461","side":"left"},{"sibling":"2abcec6d14b256f82b270a3141886de6129141dec525543607b760f02588a877","side":"right"},{"sibling":"2b6b45743f97ac502854e489ac38a3366f8ae7ede58a2728b45daadbf29e9d03","side":"right"},{"sibling":"a81babbd79ea0da9e030dd7f43bffb6519d214317decd50727bea4e78189d970","side":"right"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"8d3baa674a45fd8bc6d3e8d25298f4bec86c72fa8259d576c330d12955c7b4f7","side":"right"},{"sibling":"e32819d1eff909db08066d1703f2db3f091cddac19378b2c0c625ab11e3fdbc0","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":209569,"merkle_root":"c852efc8ad7dfffc196c71380f79e6398bcaf566974cbae7c9950b0f600d5bc8","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260603T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-03T05:37:57Z","sig_algorithm":"ed25519","signature":"1ca417607effc813341721cfdade0a2d2d4ba96a90d361dbc3e864b6991b90abc326ec41d9ba3e7110cc04adb29d7b8d95abea7ff2ab8c591b2f864ed7270c03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_3d1a7ad2700eee5b3a0607ffb9b314bf23f3d3897e9a2c58dc4e40deffd945cd"}}