{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_263670a06be47d85fd990f4105e8350a07bef134434912db926f3f9a8ad159c6","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_263670a06be47d85fd990f4105e8350a07bef134434912db926f3f9a8ad159c6","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"10f91dc2b04fd3b261140d376dd086f53016d0f7d82f1ea0890a4c69a43c8e9a","published":"Wed, 01 Jul 2026 00:00:00 -0400","receipt_hash":"10f91dc2b04fd3b261140d376dd086f53016d0f7d82f1ea0890a4c69a43c8e9a","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"10f91dc2b04fd3b261140d376dd086f53016d0f7d82f1ea0890a4c69a43c8e9a","observed_at":"2026-07-01T04:43:38.812993Z","parent_run_hash":"0e10ec7d671a375a4e18d8645653df2b4352c82fe16fae2a400af6f7634d6699","published":"Wed, 01 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2504.07385v3 Announce Type: replace-cross \nAbstract: As Large Language Models (LLMs) become increasingly used for question-answering (QA), relying on static, pre-annotated references for evaluation poses significant challenges in cost, scalability, and completeness. Meanwhile, using LLMs themselves as evaluators without external grounding remains unreliable for objective tasks, as they systematically over-accept incorrect answers, fabricate supporting rationales, and degrade sharply on questions that fall outside their training data. We propose Search-AuGmented Evaluation (SAGE), a framework to assess LLM outputs without fixed ground-truth answers. Unlike conventional metrics that compare to static references or depend solely on LLM-as-a-judge knowledge, SAGE acts as an agent that actively retrieves and synthesizes external evidence. It iteratively generates web queries, collects information, summarizes findings, and refines subsequent searches through reflection. By reducing dep","title":"SAGE: A Search-AuGmented Evaluation of Large Language Models on Free-Form QA","url":"https://arxiv.org/abs/2504.07385","vendor":"arxiv_cs_ai"},"summary":"arXiv:2504.07385v3 Announce Type: replace-cross \nAbstract: As Large Language Models (LLMs) become increasingly used for question-answering (QA), relying on static, pre-annotated references for evaluation poses significant challenges in cost, scalability, and completeness. Meanwhile, using LLMs themselves as evaluators without external grounding remains unreliable for objective tasks, as they systematically over-accept incorrect answers, fabricate supporting rationales, and degrade sharply on questions that fall outside their training data. We propose Search-AuGmented Evaluation (SAGE), a framework to assess LLM outputs without fixed ground-truth answers. Unlike conventional metrics that compare to static references or depend solely on LLM-as-a-judge knowledge, SAGE acts as an agent that actively retrieves and synthesizes external evidence. It iteratively generates web queries, collects information, summarizes findings, and refines subsequent searches through reflection. By reducing dep","title":"SAGE: A Search-AuGmented Evaluation of Large Language Models on Free-Form QA","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-01T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2504.07385"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:d72f133915d2f2a74a37fa776ccc043d8baf1f179beda1acd9a7a42f1f6f8672d1c08e8294623ba7be174805d03f07873d2d86483834d9b857086baaefa43901","signer":"crovia.substrate","subject":{"observed_at":"2026-07-01T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2504.07385"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"3b83df9c0e3bf1f53b81c4bee76c1dedd0ab4fc65b3afe59ef5dee1cbb447289","leaf_index":268665,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1a497f90e77f1890ee657bb501eb4a7190d5ada5821b2182e4ff0916f5356739","side":"left"},{"sibling":"9be6655d1d656432e4d09abd341245099af4700b44923cc8ad1e2e135a56be94","side":"right"},{"sibling":"2613bbb573e6541819134fd87bbcd0ef4bc8ca1dd789c933dc726dbf97982139","side":"right"},{"sibling":"a0362b5aa3bd32f62bed7c2d5074fa4074650bdc6e0a39fc795f164e87f38a08","side":"left"},{"sibling":"65672f56f6ed21aadf59957c0da4389950bbd60870dba031c5c71ef438a2f52e","side":"left"},{"sibling":"7caf50a8d663be3d397380d87998b2f8be0601e4f183578903035b3b15cf798b","side":"left"},{"sibling":"0eb4eb96022242f95e819116ec215f97662a006e19b7a97219e1ee83a96f4d60","side":"left"},{"sibling":"09d2631e35e497faf605540c596009624dcf181a796c7acf51592f060eb3f046","side":"right"},{"sibling":"4b14e8ccae46b588748944ffaf410222f66cebc2ac58d936cbe16847b7be9508","side":"left"},{"sibling":"bbd9a20451913e8c7b910f616d5661d9b281a94af70d201ba50b3112d429521b","side":"right"},{"sibling":"85af80e45748c1e0da1e2f42d9d66da8996016888a034f20eb17ff0a73b69eab","side":"right"},{"sibling":"4575fde969d1d9a2984cc01a37ac8441238f74527d42874272dc5582dadebb4f","side":"left"},{"sibling":"f536de281672cbf0b583a3dc46faef1de2823b72bd9265c7b60e889131cc268d","side":"left"},{"sibling":"95b8b0f67237052a17c41fa8cbdce2b79bcb5aeeba3fdb239d4c4497e598115d","side":"right"},{"sibling":"c2f351f771cee329448890504d9436792ba482e50250ca9a19289311131f96c8","side":"right"},{"sibling":"c39bfb2e911ca37ae997690bfc04128ae32e6806ee1cb3908781a7e6685022a0","side":"right"},{"sibling":"a101b4c60ef6854ac3d750eef02d8e2e06c153284b5ecb7111302f97eb129797","side":"right"},{"sibling":"eae2a3de5cb35455ad60125e196cfadaba8a590c53146e95028469f53f70349c","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":268860,"merkle_root":"d098f25810d0569730b6c0e170d57f329a70359d483b47874b5f2fc51d23de65","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260701T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-01T05:38:05Z","sig_algorithm":"ed25519","signature":"64f37f0e3df0556baa55b924a736cab8005643fa0ab65b503f22407de30eab293e6e0bd62904270fda38d05084bb350c8c0d0aadb2c5a6bae5f1c54598e6d30c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_263670a06be47d85fd990f4105e8350a07bef134434912db926f3f9a8ad159c6"}}