{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_f9e4e1c3407c7ff14939a181c94f00c0d9deeab380d1045a28900611d53b525c","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_f9e4e1c3407c7ff14939a181c94f00c0d9deeab380d1045a28900611d53b525c","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"c6f95c768c2efb4cd38cd8d0d6f46d55378a57db6ce8a650ef62fd8c3986e8a5","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"c6f95c768c2efb4cd38cd8d0d6f46d55378a57db6ce8a650ef62fd8c3986e8a5","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"c6f95c768c2efb4cd38cd8d0d6f46d55378a57db6ce8a650ef62fd8c3986e8a5","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.04034v1 Announce Type: cross \nAbstract: The language models that underpin agents have seen a rapid rise in performance on function calling benchmarks. However, the metrics used in the training and evaluation of these models often encourage models to make positive claims even when the answer is uncertain, leading to hallucinations. Such hallucinations can be disastrous when language models are trusted to use function calls to make decisions in high stakes applications. To that end, we propose an agent evaluation metric that takes into account the negative outcomes associated with incorrect function calls. Further, to catch hallucinations before they can cause harm, we propose a lightweight trainable filter that can quantify a language model's uncertainty and remove potentially harmful function calls. By training that filter to detect and suppress uncertain function calls without modifying the underlying model, we demonstrate a practical path toward agents that know when to sa","title":"The \"I Don't Know\" Filter: Enhancing Agentic Reliability in Function Calling","url":"https://arxiv.org/abs/2607.04034","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.04034v1 Announce Type: cross \nAbstract: The language models that underpin agents have seen a rapid rise in performance on function calling benchmarks. However, the metrics used in the training and evaluation of these models often encourage models to make positive claims even when the answer is uncertain, leading to hallucinations. Such hallucinations can be disastrous when language models are trusted to use function calls to make decisions in high stakes applications. To that end, we propose an agent evaluation metric that takes into account the negative outcomes associated with incorrect function calls. Further, to catch hallucinations before they can cause harm, we propose a lightweight trainable filter that can quantify a language model's uncertainty and remove potentially harmful function calls. By training that filter to detect and suppress uncertain function calls without modifying the underlying model, we demonstrate a practical path toward agents that know when to sa","title":"The \"I Don't Know\" Filter: Enhancing Agentic Reliability in Function Calling","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.04034"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:182374bd4524b3956e8766611164f4345de788c1dc77ebd805ab5f3ca1de4b0af3c1b013bbcefb4e8a4a3dd9ea74918574789b60163d835a69a4b92945af2b0b","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.04034"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d0056f78fdd8ce9f0345a43553dc4985e8fa99b3cfc4bb0ec6c91fbab36e70c2","leaf_index":289010,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"b60155a9c21faba338bf940a2b2c20b0eb887fe41898c8730d3c02981df8a01d","side":"right"},{"sibling":"021052745f95157014664f95cd029d5f0e5ee85a6555e8fff6b689e52f401e3b","side":"left"},{"sibling":"dcd66ce1b92a82fd972a977057a02c2e5d6ce490e3296c9866c10819d5ceb884","side":"right"},{"sibling":"2bbe7ae8b1ea4b9d1424a129832058389eb2382e56858f3537e877751f5db169","side":"right"},{"sibling":"8337f0062fac07c60bb0a77810696db0954b008ee9c9d00435dde49419c7335a","side":"left"},{"sibling":"3738e35c17cf15c3db999b27b37e65c9d5e33de0f39dbf34df98a1cfdb4c16cb","side":"left"},{"sibling":"50a7913417ce856965a08aaadd04ae045e1d09e9d1ce0c00f8df88f007185c80","side":"left"},{"sibling":"e2f3465fc74f48bb335c248143b3811c6e2bffa1b6671301e8a48cb5569a68db","side":"left"},{"sibling":"1fe4f74c7ddb4bf1f3105cee6c2413b68972083840bc89ce8e21eddb44153505","side":"right"},{"sibling":"841129c21a7583176cdc7de281cadfe0e00d04673461e199760cd5128d8cc2d5","side":"right"},{"sibling":"19d6dfd29bc47f35fa02e8fe765277ba9cc3e6da5072309f24ebaac5b5f295e3","side":"right"},{"sibling":"8e0ad7889eb2d4b40e5b6c3d8e2eb19d4e202374983f468aa76321823de07a9f","side":"left"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_f9e4e1c3407c7ff14939a181c94f00c0d9deeab380d1045a28900611d53b525c"}}