{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1af2a8f7f795825192b2521ad428f970dad439fe9aefd69b3e36828a82cbb403","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1af2a8f7f795825192b2521ad428f970dad439fe9aefd69b3e36828a82cbb403","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"ba569c2fe4f5dc762110446e8ac1b6f6149dc297b3f56395da77d4ce0afea876","published":"Fri, 26 Jun 2026 00:00:00 -0400","receipt_hash":"ba569c2fe4f5dc762110446e8ac1b6f6149dc297b3f56395da77d4ce0afea876","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"ba569c2fe4f5dc762110446e8ac1b6f6149dc297b3f56395da77d4ce0afea876","observed_at":"2026-06-26T04:43:58.168958Z","parent_run_hash":"9459505a803125e4b968df08c74ed0054a2aafd44e4e1a036e3b0709a8a65cb4","published":"Fri, 26 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.26101v1 Announce Type: cross \nAbstract: Reliable evaluation of large language models should separate supported answering from unsupported guessing without conflating either with data contamination, prompt idiosyncrasy, or generic refusal behavior. We present a contamination-aware, multi-zone benchmark for measuring the transition from answerable knowledge to abstention-expected unknowns under frozen build-time labels. The benchmark contains 1,200 items across five domains, explicit abstention expectations, contamination-risk metadata, and dual parsing with an official strict parser plus a normalized robustness parser. We evaluate FLAN-T5, Qwen2.5-Instruct, and Llama-3-Instruct models under locked answer-or-abstain prompts, answer-only controls, and prompt-template variants. The benchmark is not solved by generic non-answer behavior: FLAN baselines remain weak on productive abstention, while stronger instruction-tuned models expose a selective but incomplete transition from a","title":"Know2Guess: A Contamination-Aware Multi-Zone Benchmark for Knowledge-Boundary Evaluation in Large Language Models","url":"https://arxiv.org/abs/2606.26101","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.26101v1 Announce Type: cross \nAbstract: Reliable evaluation of large language models should separate supported answering from unsupported guessing without conflating either with data contamination, prompt idiosyncrasy, or generic refusal behavior. We present a contamination-aware, multi-zone benchmark for measuring the transition from answerable knowledge to abstention-expected unknowns under frozen build-time labels. The benchmark contains 1,200 items across five domains, explicit abstention expectations, contamination-risk metadata, and dual parsing with an official strict parser plus a normalized robustness parser. We evaluate FLAN-T5, Qwen2.5-Instruct, and Llama-3-Instruct models under locked answer-or-abstain prompts, answer-only controls, and prompt-template variants. The benchmark is not solved by generic non-answer behavior: FLAN baselines remain weak on productive abstention, while stronger instruction-tuned models expose a selective but incomplete transition from a","title":"Know2Guess: A Contamination-Aware Multi-Zone Benchmark for Knowledge-Boundary Evaluation in Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-26T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.26101"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:8f548eb7853684efde7db67018f88702dbceeb1998ad1b5c854e8f91c5afce041eda91f1ee5392d1cc21ad8b1272fcdb31a475316508ab6c3bf486dea2ba3c02","signer":"crovia.substrate","subject":{"observed_at":"2026-06-26T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.26101"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"68a29fc00e353e2eba302f65d6d7f13f3e6fd8b6eb528d7dc8b4dbc1fe69630f","leaf_index":251063,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"b7667032e792ad301f7d0622555e40fce1b0efeaf936aeecc9e66a1557ed8cf9","side":"left"},{"sibling":"e71812a14f19f1efdad39c70ca4860daea34e0e0ec4c59eee5ed7f483500d1d8","side":"left"},{"sibling":"83abfbb5aa68ba3488e387d68b6c446b6fda61e0b0299ef29a61c13f053d3f91","side":"left"},{"sibling":"170569eaded236021d6c02d463c5c17f4379fcdf13f10e090ee2a599ab4c1fab","side":"right"},{"sibling":"ed68287d1915ab292e39acf04af5a28ed8f162b77ec2d989b76a53b85aa53d20","side":"left"},{"sibling":"78203b2e2864794a122eaedd882440e0ed445673a2d3d6166066180ea87f64dc","side":"left"},{"sibling":"b31fd8919314f43e5528d32d692e1d0c9345f86173f69bd11bcf95d27af3147a","side":"right"},{"sibling":"762a50202e5bb846bfb5afec2f3cc80d9313546598c6c53f53acc51c6325b763","side":"left"},{"sibling":"e276c2896e885a069398e8350a2d9aae49ed0ab34771e352045341312283b40a","side":"right"},{"sibling":"b6709caadc8510310ee2ec0b66d1058fcad31c91c65cc2bad6f46a693d553580","side":"right"},{"sibling":"a72c3b8804a37d1a9d18e02e6fdb048bc8ea6b0746909cd2de10bdbabc793737","side":"left"},{"sibling":"803703dc2c50a646fa77b57c0056e9a5126611ba4bcde0d6013ccd6b2d44bdbf","side":"right"},{"sibling":"e78f244b1b8df6d5e3fdc6dd76b5c27d4e6fe3b93b8cd61355497b63d7e4cfe8","side":"left"},{"sibling":"6167cb552ed6871fbf0afcf3db01d1017af7d472b136fcbe5404d1df09f41cc1","side":"right"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":251380,"merkle_root":"e042805d07cd8dc777d49695ad78b8d4ec9721df271ff0245d7706773c30b4a5","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260626T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-26T05:37:59Z","sig_algorithm":"ed25519","signature":"c68a6e827804771acd244208495c5e35c6f417307db3c76192f038dedaa5b5f86e019074a3bea353357ff7457924c4f7907832638bb9a4d3d18e7a7f50c6a10b","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1af2a8f7f795825192b2521ad428f970dad439fe9aefd69b3e36828a82cbb403"}}