{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_5528d9bafd7ded62d0b03edb1b59caea8b225ea2830b07f3f9291eb58e80aefe","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_5528d9bafd7ded62d0b03edb1b59caea8b225ea2830b07f3f9291eb58e80aefe","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"bc8e0bbd12c838baad464324b36fa427895953e44795f453da004395cd6e0f09","published":"Tue, 14 Jul 2026 00:00:00 -0400","receipt_hash":"bc8e0bbd12c838baad464324b36fa427895953e44795f453da004395cd6e0f09","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"bc8e0bbd12c838baad464324b36fa427895953e44795f453da004395cd6e0f09","observed_at":"2026-07-14T04:43:37.834979Z","parent_run_hash":"66b89520a448b8d9fe7d8f602ef38b82b6c95e57f532ce72de51375b41870477","published":"Tue, 14 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.10059v1 Announce Type: new \nAbstract: Agent systems based on large language models (LLMs) are increasingly deployed for autonomous tasks, yet existing evaluations mostly focus on task success rather than whether agents know when to abstain. This gap poses real risks: under ambiguity, conflicting constraints, or tool failures, agents may execute unintended and irreversible actions. To close this gap, we present the first systematic evaluation framework for agentic abstention: the calibrated ability of tool-using LLM agents to recognize when not to act. At its core, AgentAbstain is a paired-task benchmark built on an agent-native taxonomy of 8 abstention scenarios across pre-execution reasoning and runtime discovery. It contains 263 paired tasks across 42 executable sandbox environments, where each pair consists of a should-act task and a should-abstain variant produced through a controlled perturbation to the instruction, tool, or environment state. To scale this paired desig","title":"AgentAbstain: Do LLM Agents Know When Not to Act?","url":"https://arxiv.org/abs/2607.10059","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.10059v1 Announce Type: new \nAbstract: Agent systems based on large language models (LLMs) are increasingly deployed for autonomous tasks, yet existing evaluations mostly focus on task success rather than whether agents know when to abstain. This gap poses real risks: under ambiguity, conflicting constraints, or tool failures, agents may execute unintended and irreversible actions. To close this gap, we present the first systematic evaluation framework for agentic abstention: the calibrated ability of tool-using LLM agents to recognize when not to act. At its core, AgentAbstain is a paired-task benchmark built on an agent-native taxonomy of 8 abstention scenarios across pre-execution reasoning and runtime discovery. It contains 263 paired tasks across 42 executable sandbox environments, where each pair consists of a should-act task and a should-abstain variant produced through a controlled perturbation to the instruction, tool, or environment state. To scale this paired desig","title":"AgentAbstain: Do LLM Agents Know When Not to Act?","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-14T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.10059"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ef0f7123da91a36bac0b07a6a03aa082db82824400dc85965008c30fdcd0144b82ce3330e645bea153a20f13a7eb7c40c61f68c599caa29e49b32d8b7556710e","signer":"crovia.substrate","subject":{"observed_at":"2026-07-14T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.10059"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"540f66a3450c4fe4d94b75fed82d096a8cdab7f124691db85a94f781cba631f9","leaf_index":312813,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7ee3a73cf6a82b60ef0b784fd49d1e051bebc5d347c6144077ddc2c1c9ce54bd","side":"left"},{"sibling":"61f6afcbda96613231055f67d5c5cdd0185e4a866eb2acc4f6f1c46fe3ac2087","side":"right"},{"sibling":"6172e032cd8cc81e985f1f05aaf6287338aadbdc46e37a2f6332664cf6d78894","side":"left"},{"sibling":"550543ad3684e75bc51a9da058a28c956872b055421b6815b43f3c1cdcd51756","side":"left"},{"sibling":"e6b51982312a070a32dee7d04c7978a0c6663fbf33e2d486b1b9a3041200fbba","side":"right"},{"sibling":"bc4721fdcc1f36bedeaba62d3f41fc689ea8d373f9d3d57a988797bbcd8257ae","side":"left"},{"sibling":"e78d0d4e476fa207968b0d1b4de1b6e0b862fa2573890cb17883fbd17c4bee92","side":"left"},{"sibling":"1bf33ad3b3babcd7e3971fa58862cd0ad2723ec6c97ea2f6ab19b733299eb475","side":"left"},{"sibling":"36e58162e2f6481e200fac2cc34bef53d1ba911d1fe07a8ecd0e0a03097469fc","side":"left"},{"sibling":"2b62f9c6ec45961991e3cfb6ed140d70943061d83fb37c4dc886775dd52fcb6a","side":"right"},{"sibling":"1627ce6908965f2e5fbc7c16bc8e247867478c587014561d0de1359dc2dc0afc","side":"left"},{"sibling":"74e1ba9fd48bdd0f52777dd5b56c10706e73f42629013748eca13ef113cd59db","side":"right"},{"sibling":"682fd39539db5bce908d0ab6f180374728fed6b34a9a8c87c7b4eb5a726e30b8","side":"right"},{"sibling":"184927f71d667ac0c6ebeadb33394626973738b9107c6dc3c0c4949b44acf295","side":"right"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"c4c189d79669979b59d9aa976ee249c54a7e065b4a05bf49463459cc7f37428e","side":"right"},{"sibling":"19475e206bdf2698769a286db2c97c4d3f089741319f9b29a139038c6e511975","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":313381,"merkle_root":"5f7d48580373d0ab0bc6bdf34f26de442f9a86d130946f7ee45addd1a55cb9f4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260714T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-14T05:38:23Z","sig_algorithm":"ed25519","signature":"97364224142ed1b2a8925fed1bfba2e0a8a6a15f8ca0999b934846ad83584a78c7f0f5cd09dbc1c938b35383fcc5c8fa4b723118e30916628879d1c5fd3a9908","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_5528d9bafd7ded62d0b03edb1b59caea8b225ea2830b07f3f9291eb58e80aefe"}}