{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_473ef0cf040df5ed9dbe4356a77d26c348baa6c87cb8f89644c335da2a19dedd","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_473ef0cf040df5ed9dbe4356a77d26c348baa6c87cb8f89644c335da2a19dedd","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"eec45884c831d205ee0f25ab91f61be28213847fb4f2671f50f8027b6389218b","published":"Tue, 28 Jul 2026 00:00:00 -0400","receipt_hash":"eec45884c831d205ee0f25ab91f61be28213847fb4f2671f50f8027b6389218b","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"eec45884c831d205ee0f25ab91f61be28213847fb4f2671f50f8027b6389218b","observed_at":"2026-07-28T04:43:08.282317Z","parent_run_hash":"23a1ef85134515049ced29518443d084afc46fd7c967741e6c6acdbdbbf29939","published":"Tue, 28 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.22951v1 Announce Type: cross \nAbstract: Reliability assessment of large language models (LLMs) seeks to estimate the probability that a model produces correct responses under a specified operational profile. Conventional benchmark-based evaluation, often summarized by aggregate accuracy, provides a point estimate of performance but does not characterize the uncertainty associated with reliability claims. Currently, statistical inference methods for LLM reliability assessment are emerging. However, a key assumption underlying these models is that test outcomes can be treated as independent repeated trials. This assumption may be inappropriate in sequential settings, where later responses depend on earlier interactions through retained context, error propagation, or an evolving interaction state. We extend a hierarchical Bayesian framework for LLM reliability assessment by relaxing the assumption of independent task outcomes and introducing a Hidden Markov Model to capture seq","title":"Modeling Memory-Dependent Reliability of LLMs: A Hidden Markov Model","url":"https://arxiv.org/abs/2607.22951","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.22951v1 Announce Type: cross \nAbstract: Reliability assessment of large language models (LLMs) seeks to estimate the probability that a model produces correct responses under a specified operational profile. Conventional benchmark-based evaluation, often summarized by aggregate accuracy, provides a point estimate of performance but does not characterize the uncertainty associated with reliability claims. Currently, statistical inference methods for LLM reliability assessment are emerging. However, a key assumption underlying these models is that test outcomes can be treated as independent repeated trials. This assumption may be inappropriate in sequential settings, where later responses depend on earlier interactions through retained context, error propagation, or an evolving interaction state. We extend a hierarchical Bayesian framework for LLM reliability assessment by relaxing the assumption of independent task outcomes and introducing a Hidden Markov Model to capture seq","title":"Modeling Memory-Dependent Reliability of LLMs: A Hidden Markov Model","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-28T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.22951"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:0336fb6378f9a35a2b20a15b58d28c14422af06c2e9f5666c4103bb84441dd58c0ea978cd46a0bd6a5f2731c1db8f20a7d52f5660694843d192edf759494f602","signer":"crovia.substrate","subject":{"observed_at":"2026-07-28T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.22951"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"0b1239d87f05de7831090d9c32231d4929820639bd139ef5167ce1869b8d7905","leaf_index":360556,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"2595d1448cfe1f8d2bad0939dede81fa8c8fa6e49aeab73acd189220819b7f7a","side":"right"},{"sibling":"9985cbe21e817167d58274bbaa045a46099f9b6f3e347f5b1267bcaa37a630d2","side":"right"},{"sibling":"f2c5e0bccc24991c7084ffe987a71cf09cefd6af5c37df00c831ca1f462ab8c9","side":"left"},{"sibling":"ad353dc30c7d4d20d0146facd99d29d6caa9ed149e2ca771627fa6ab342251ae","side":"left"},{"sibling":"48bde4008179952c976543ce73cd688c2ee13ea83cc2d133aebc3165e6d92c55","side":"right"},{"sibling":"683739b49831eb9d8ed7ad240c7843a793d04a14739fa08c2deb614e19a4e9d8","side":"left"},{"sibling":"a75ea3fe312c1d9ab3776bfe70acd45752855705bf7bb0ab8e19b5f21d865d19","side":"left"},{"sibling":"7da15787b28ee42d6f6ffa689ccb25f3c7958b4ad112266a2a4acad4347e0386","side":"right"},{"sibling":"eb79f1d599c07786d2268140481af8b617999ecc5688c6283fe58e5e6f3a2640","side":"right"},{"sibling":"fe03c6d0b083c4097049d7fc6d7d08dfdb05d9c198b2786336689712d66eabf0","side":"right"},{"sibling":"590ea76fdfc1b9e8072055c378be3182f916709e8d06bd955232e650a7188b86","side":"right"},{"sibling":"d92781c59301ffd5bfb0bad75d9fdf6d73879f715518149d383f7362213e0daf","side":"right"},{"sibling":"f315303d4402b57497416d48eb4c4bb50405b40862d41c7caf318cb3d29c5237","side":"right"},{"sibling":"36973eb5f586cd67e0c0dc055dd87e734aa544c35d4400d2c9f932f8ef8fb27f","side":"right"},{"sibling":"e39f7900355489c4718b21cc2d3d06382e1d2f22b864d10be9ebc9d498b279a1","side":"right"},{"sibling":"1f9a970b25dd938c98cabc9e0a55c5a6f46292b9fa37cc89d90ef0cbb1e05a8c","side":"left"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"3b50864499c874394ea0928567747666eaf59b01380e46cd52164ec5acec0f71","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":361008,"merkle_root":"3065e8369ea437c06beba806dc4e4bb159979adeb21fe632242c1906a7204647","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260728T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-28T05:38:48Z","sig_algorithm":"ed25519","signature":"9141644407577a82611c1579110f667de2d46dc6b93c6322edf26f4c3056ea99f0e56502853908e30d87c38bcf95eb6e0ab5130525aa51505bd6f61938120609","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_473ef0cf040df5ed9dbe4356a77d26c348baa6c87cb8f89644c335da2a19dedd"}}