{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_db62b6a111f366e7ad2d79228e4c625e7289069126fbc47f3e676acc0be95da3","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_db62b6a111f366e7ad2d79228e4c625e7289069126fbc47f3e676acc0be95da3","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"1e5d469f694e5d838307f0737eeea8d51f73c362d2712ba493c02b6f6694c1b7","published":"Thu, 11 Jun 2026 00:00:00 -0400","receipt_hash":"1e5d469f694e5d838307f0737eeea8d51f73c362d2712ba493c02b6f6694c1b7","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"1e5d469f694e5d838307f0737eeea8d51f73c362d2712ba493c02b6f6694c1b7","observed_at":"2026-06-11T04:43:37.662146Z","parent_run_hash":"5267801b61ae0d882196b5f37208a9a1633905a64ca7d933f1fa5075cd861491","published":"Thu, 11 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.28591v2 Announce Type: replace-cross \nAbstract: The validity of AI safety evaluations depends on models behaving consistently across controlled and deployment settings. Prior work has identified test-time contextual cues, such as hypothetical scenarios, as a source of verbalized evaluation awareness and subsequent behavioral shift. In this paper, we investigate a potential explanation of this phenomenon: evaluation meta-knowledge, defined as parametric knowledge about the structural traits that characterize evaluations. Similar to dataset contamination, where benchmark exposure leads to higher performance through memorization, we hypothesize that models trained on texts describing evaluation practices may implicitly learn to recognize and respond to evaluation-like contexts, for instance, through exposure to scientific articles or social media posts about AI benchmarking. To test this, we fine-tune models on synthetic documents describing evaluation traits such as verifiable","title":"Models That Know How Evaluations Are Designed Score Safer","url":"https://arxiv.org/abs/2605.28591","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.28591v2 Announce Type: replace-cross \nAbstract: The validity of AI safety evaluations depends on models behaving consistently across controlled and deployment settings. Prior work has identified test-time contextual cues, such as hypothetical scenarios, as a source of verbalized evaluation awareness and subsequent behavioral shift. In this paper, we investigate a potential explanation of this phenomenon: evaluation meta-knowledge, defined as parametric knowledge about the structural traits that characterize evaluations. Similar to dataset contamination, where benchmark exposure leads to higher performance through memorization, we hypothesize that models trained on texts describing evaluation practices may implicitly learn to recognize and respond to evaluation-like contexts, for instance, through exposure to scientific articles or social media posts about AI benchmarking. To test this, we fine-tune models on synthetic documents describing evaluation traits such as verifiable","title":"Models That Know How Evaluations Are Designed Score Safer","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-11T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.28591"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ea70b23d1913f4dd15318e82d9a2de72b87fb8ba3a5284b02f221cb145d42198557d3859dbbd28bab06eacfa86d28ef55caeebc877cbe9e7296bf904098e1b07","signer":"crovia.substrate","subject":{"observed_at":"2026-06-11T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.28591"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"0b6f49634abe96a296801f6a9978c34dd0f75475c552dfedd32ccd4578698092","leaf_index":227630,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"478a8ffbc429f1e2aac2a65317de29ebb6a884495339aa0e72aa63dd16a55551","side":"right"},{"sibling":"925f22e3a9bd3932c881f40f82752b6d6fad4389a1a7c326f985e6e5cab58057","side":"left"},{"sibling":"5820502d28f79ac7e65e6a741a2975ba5babc508c3aef1261365dfe7a1d4c38c","side":"left"},{"sibling":"ef822b8bfb2a0c312adf6d433bb41c96f38d5f6bcaf8f153a203d34e083546ba","side":"left"},{"sibling":"696baab52c8dd7da00eec185cbbf6d57c893178092dfa3f22b70aabd97d81169","side":"right"},{"sibling":"bf0a852b6519918429c7e01817ee3e63de5ebb9c3bf1960b2f07a5135d3a4194","side":"left"},{"sibling":"b3e44c169468247ff310527cf2f90f2c6c7fb8fa20f54d288a0ad8da5e1aeaae","side":"right"},{"sibling":"8a3e5c7ea49dd66cd71ed0760cc762c1e6d233215488abe34cb40cd8d01bc326","side":"right"},{"sibling":"f76de2250b6d1132375cb2a9c157faf5c22b74a7ca5c8df342d778d0e08e1216","side":"left"},{"sibling":"04e399458c5b36988cae0bf1c6dbe1b01349003b15cb5aa43f95c55acffe4ec3","side":"right"},{"sibling":"1383228337d54218bd8e5563aebb0b0dfe15c5269e3d5138e64c261d6130a88b","side":"right"},{"sibling":"57cb49c192550231071a0bf53a0821da2f79c585ec6c8d0fc76cebd62ccd78b2","side":"left"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_db62b6a111f366e7ad2d79228e4c625e7289069126fbc47f3e676acc0be95da3"}}