{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_4b5c84f87232c20354d2f2240a9e01f550943fadefe5f1ef8c42abe8ca80de5d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_4b5c84f87232c20354d2f2240a9e01f550943fadefe5f1ef8c42abe8ca80de5d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"3c9b8ee86eea9f3ac2f86d7f9d0082a9ceb14a786abbd8e479d3744348d5f173","published":"Thu, 09 Jul 2026 00:00:00 -0400","receipt_hash":"3c9b8ee86eea9f3ac2f86d7f9d0082a9ceb14a786abbd8e479d3744348d5f173","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"3c9b8ee86eea9f3ac2f86d7f9d0082a9ceb14a786abbd8e479d3744348d5f173","observed_at":"2026-07-09T04:43:38.345231Z","parent_run_hash":"3e22c7c40abc4d94232acf1766a43492b8b8d51d10a58f1109535988a16554e6","published":"Thu, 09 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.07184v1 Announce Type: cross \nAbstract: Pre-deployment safety evaluations aim to inform the downstream risks of releasing a new AI model. Yet most evaluations provide limited evidence about how often undesired model behavior will occur in deployment: they generally have insufficient coverage, are unrepresentative, and are generally recognizable as tests. To address these concerns, we study a simple way to simulate a model deployment: starting from de-identified conversations from a previous model deployment, we hold fixed the initial conversation prefix and regenerate the next response using a candidate model. The resulting responses can then both be audited for novel misalignments and used to estimate the prevalence of model misbehavior before deployment. We evaluate deployment simulation across four GPT-5-series deployments, using registered, outcome-blinded predictions for GPT-5.4 and retrospective analyses of three earlier releases. We find that deployment simulation pro","title":"Predicting LLM Safety Before Release by Simulating Deployment","url":"https://arxiv.org/abs/2607.07184","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.07184v1 Announce Type: cross \nAbstract: Pre-deployment safety evaluations aim to inform the downstream risks of releasing a new AI model. Yet most evaluations provide limited evidence about how often undesired model behavior will occur in deployment: they generally have insufficient coverage, are unrepresentative, and are generally recognizable as tests. To address these concerns, we study a simple way to simulate a model deployment: starting from de-identified conversations from a previous model deployment, we hold fixed the initial conversation prefix and regenerate the next response using a candidate model. The resulting responses can then both be audited for novel misalignments and used to estimate the prevalence of model misbehavior before deployment. We evaluate deployment simulation across four GPT-5-series deployments, using registered, outcome-blinded predictions for GPT-5.4 and retrospective analyses of three earlier releases. We find that deployment simulation pro","title":"Predicting LLM Safety Before Release by Simulating Deployment","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-09T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.07184"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:0accde2334ca17e61347afa71c591fdecce73057114b03d1d62fb23746cfd4dad4e6467c6a0c03ed3b1f6ade1a528fdef7a457b86adf1584e6a5a22fe97e3c0d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-09T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.07184"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"746abe1733d848dc1ca77b372a6c754fe41e3745f66f69791a4f5f88d3280ff6","leaf_index":296112,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"dc1c908cb136c4d956d08f60c95e16d1745fdc5cd149af54fb316a0b48f2672e","side":"right"},{"sibling":"6124b182f7565c56bba82d68ca415fde8971a4304b3b6beed331f6a2f1a3e00b","side":"right"},{"sibling":"b0a3f9df70558ab76d3cbdbc55f6701d4b7980d3928ae7852627c2cc7b2d3923","side":"right"},{"sibling":"180c4a704c34916e76b6315ad8ec13a83be5a4b13ff794d9dc336cef9bec08e5","side":"right"},{"sibling":"eaffbb28019c0675fc1263ceee44e6b2d0e1ccd4238bc612614d6a8a30e3937c","side":"left"},{"sibling":"81f21ad902302e2af532f5380ae8c3ffee9daaa01b6c1f3a552ca48a362960fa","side":"left"},{"sibling":"6c45b7ac79d3163d45e1d56cd7db0b01778b873ed657ef89d1d09b6352b1b91e","side":"right"},{"sibling":"6e8f2e16cb75beb661e7f7b63be19804b6f6474dfbf899f8e417b19695f2fba4","side":"left"},{"sibling":"47a94dfb6e50a020e68582e6af2e6a8cf5a4c7efe9fed3deaedbc51f53d75367","side":"right"},{"sibling":"92bb57de69c78fd32ac7108b10d81676c184265a5a53de3c4b22d8cf3b54b499","side":"right"},{"sibling":"85a226efd14acc17835b04bc26706fa44595edbd531f194faf59f60ab72d4bb8","side":"left"},{"sibling":"da38b05536b12aee196b6ac988739211c257d32da790faccf5ac4b0cbc1bb15c","side":"right"},{"sibling":"d438dc3eddb0b14dc8b97cd021a4f44545ce5a8e827f3ea4044fd32b1877475e","side":"right"},{"sibling":"f73ad10346837ae47f59f0647f79b9416e1d499bf2b90af44448e7f09372200a","side":"right"},{"sibling":"bdc09902fcd434c0f7d3e680bf550e560777228c0b085ce80c637ce97fc4104c","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"ba603dffe985ef518e3a72793a3eaca83a7f1a79e5491fd0f62f421339f2d137","side":"right"},{"sibling":"be20b90931f0a14e3558ea4387537200fcbd14e019b3c5ed07a2ae4c62fc7c42","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":296360,"merkle_root":"64af62f723a5bc02adfa98b77e2006fc634de4ebf68626694f052342a200bea2","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260709T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-09T05:38:18Z","sig_algorithm":"ed25519","signature":"92ece7411e0d82898aac164e7d6573a6d0f7a595aad0780d710d873e548a061d2a8678cef4d237d3bf0eabe0a5f766b41cbc0b4bada2801e7532026291b4a309","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_4b5c84f87232c20354d2f2240a9e01f550943fadefe5f1ef8c42abe8ca80de5d"}}