{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_744cc425ebb2c3b4c3dd560a75157a61d842555d80ec824c791405d4ec71ff51","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_744cc425ebb2c3b4c3dd560a75157a61d842555d80ec824c791405d4ec71ff51","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"68b9dd2a778ab75f4168a7f71997c0220dfb9b3ae8abab820ab413259a130df8","published":"Tue, 28 Jul 2026 00:00:00 -0400","receipt_hash":"68b9dd2a778ab75f4168a7f71997c0220dfb9b3ae8abab820ab413259a130df8","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"68b9dd2a778ab75f4168a7f71997c0220dfb9b3ae8abab820ab413259a130df8","observed_at":"2026-07-28T04:43:08.282317Z","parent_run_hash":"23a1ef85134515049ced29518443d084afc46fd7c967741e6c6acdbdbbf29939","published":"Tue, 28 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.22554v1 Announce Type: new \nAbstract: Large language models (LLMs) often achieve strong accuracy on benchmarks, yet it remains unclear how reliably they apply this knowledge when the same question is phrased in different but equivalent ways. In this work, we study how model answers change under meaning-preserving paraphrases across factual question answering and mathematical reasoning tasks. Across four benchmarks and 13 models, we find that model outputs frequently depend on the exact wording of the prompt. While overall accuracy typically changes only modestly across paraphrases, instance-level behavior is far less stable: for many questions, models alternate between correct and incorrect answers depending on phrasing, with mismatch rates reaching more than 23%. Conditioning on questions that are answered correctly in their original form reveals even larger failures measured by answer flip rates, showing that single-prompt correctness is often a poor indicator of reliabili","title":"Same Question, Different Answers: Evaluating LLM Reliability Beyond Accuracy","url":"https://arxiv.org/abs/2607.22554","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.22554v1 Announce Type: new \nAbstract: Large language models (LLMs) often achieve strong accuracy on benchmarks, yet it remains unclear how reliably they apply this knowledge when the same question is phrased in different but equivalent ways. In this work, we study how model answers change under meaning-preserving paraphrases across factual question answering and mathematical reasoning tasks. Across four benchmarks and 13 models, we find that model outputs frequently depend on the exact wording of the prompt. While overall accuracy typically changes only modestly across paraphrases, instance-level behavior is far less stable: for many questions, models alternate between correct and incorrect answers depending on phrasing, with mismatch rates reaching more than 23%. Conditioning on questions that are answered correctly in their original form reveals even larger failures measured by answer flip rates, showing that single-prompt correctness is often a poor indicator of reliabili","title":"Same Question, Different Answers: Evaluating LLM Reliability Beyond Accuracy","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-28T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.22554"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:6b0ef015d0623a14963848ac91d95e5ba49daccb2c2cd861b804aa4f15e87a3d050119f13d6c9e82e79c521276affe7a1acb979db8012b97e651548fb06fc00c","signer":"crovia.substrate","subject":{"observed_at":"2026-07-28T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.22554"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"fa006113b264ad68fef639b3bab6ac89e53ee75e8853e97769ca8dfe868a56ae","leaf_index":360310,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"4407121bf535c308f81f85f53354860d8c16c78d6bcc3d13eb968881ac0b6c88","side":"right"},{"sibling":"69ad6910cce757a5013bc96c770638863ab2039a12b7cb2519c04e76c9e2e4c3","side":"left"},{"sibling":"c4a2cf78624993be9f31399d11d7220daef98040b2b8ecd43b169ecc4f0b1983","side":"left"},{"sibling":"1c6246aab93bc4404e9870cecc23a4722289bbc2e6e6623ed5c953702d279e7d","side":"right"},{"sibling":"e643940e38743722dc8db0e8ac3f77e9f37deb207f4fe0be0e98557ff117ea8f","side":"left"},{"sibling":"f684cd134aec2f15eab82b0a08e9fa0a557907ac0bc0e8253d580372c2d032a8","side":"left"},{"sibling":"38472831c334dec7e68705f1cb832b1fc560a14ec1da8efd7dfa9f97dc706493","side":"left"},{"sibling":"974d4082909beffa6d7500ec421094d0b0636c1568e2f740bd6a77f88543d165","side":"right"},{"sibling":"8fe957ea378915d410087f8b2c41171bdfdac25ba34c42d93c5aaa835f32246c","side":"left"},{"sibling":"f57ebea8749a327b4323e9ad9c906a9400f5a5d8c83084dfc7d80be83bd44007","side":"left"},{"sibling":"4424ff61f20c9cef2251672611ec86369d87b1608ab089738930fd3d52e0dd54","side":"left"},{"sibling":"8db22b6d8b4004df8ccd40f648d9fd8a62f564821c80fbfd9c743849f4102e1b","side":"left"},{"sibling":"cff500acb83b14a8a7195a72d90dd4b7a6b8a28fc1069f8300b1191728718f90","side":"left"},{"sibling":"51ee2566d84aba54b9d07233fb460f9fab044b4edfa3fbe376fa1df0727acfa5","side":"left"},{"sibling":"f3e45bceed774d2402fa45d41ff5190f295823bd2f216eb90157884150034693","side":"left"},{"sibling":"2096cd69b54e283ddc45b26b63564a5e4c02f303ffd855b3b3533bcf26bc284d","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"3b50864499c874394ea0928567747666eaf59b01380e46cd52164ec5acec0f71","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":361008,"merkle_root":"3065e8369ea437c06beba806dc4e4bb159979adeb21fe632242c1906a7204647","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260728T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-28T05:38:48Z","sig_algorithm":"ed25519","signature":"9141644407577a82611c1579110f667de2d46dc6b93c6322edf26f4c3056ea99f0e56502853908e30d87c38bcf95eb6e0ab5130525aa51505bd6f61938120609","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_744cc425ebb2c3b4c3dd560a75157a61d842555d80ec824c791405d4ec71ff51"}}