{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_af132954eb3a9589c6f47dce441cd2ab801f6adebe20b67e834bd5f9f5ec9062","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_af132954eb3a9589c6f47dce441cd2ab801f6adebe20b67e834bd5f9f5ec9062","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"5a2ea4510d51bbfa68a1d4704d9bfca5e043402e3a651e5b28693ffbb72e0d14","published":"Thu, 23 Jul 2026 00:00:00 -0400","receipt_hash":"5a2ea4510d51bbfa68a1d4704d9bfca5e043402e3a651e5b28693ffbb72e0d14","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"5a2ea4510d51bbfa68a1d4704d9bfca5e043402e3a651e5b28693ffbb72e0d14","observed_at":"2026-07-23T04:43:18.446221Z","parent_run_hash":"b3f5e4095688e31d15c25de2607bca42111343bdfdac967d48e47905415ddea0","published":"Thu, 23 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2512.05518v2 Announce Type: replace-cross \nAbstract: Open-source Large Language Models (LLMs) play a critical role in the democratization of AI, yet their \"open\" nature introduces more avenues for malicious actors to misuse them for harmful purposes. A frustratingly easy but powerful technique known as the prefilling attack has been shown to effectively circumvent the safety alignment of frontier open-source LLMs by simply prefilling the assistant response with an affirmative prefix before decoding. A recent promising supervised fine-tuning defense proposes using a simple data augmentation scheme to achieve a \"deep\" safety alignment, allowing the model to generate natural language refusals immediately following harmful prefills. In this work, we show that a simple generalization of the prefilling attack, which we refer to as the Rank-Assisted Prefilling (RAP) attack, can effectively extract harmful content from models fine-tuned with the data augmentation defense by selecting low","title":"Matching Ranks Over Probability Yields Truly Deep Safety Alignment","url":"https://arxiv.org/abs/2512.05518","vendor":"arxiv_cs_ai"},"summary":"arXiv:2512.05518v2 Announce Type: replace-cross \nAbstract: Open-source Large Language Models (LLMs) play a critical role in the democratization of AI, yet their \"open\" nature introduces more avenues for malicious actors to misuse them for harmful purposes. A frustratingly easy but powerful technique known as the prefilling attack has been shown to effectively circumvent the safety alignment of frontier open-source LLMs by simply prefilling the assistant response with an affirmative prefix before decoding. A recent promising supervised fine-tuning defense proposes using a simple data augmentation scheme to achieve a \"deep\" safety alignment, allowing the model to generate natural language refusals immediately following harmful prefills. In this work, we show that a simple generalization of the prefilling attack, which we refer to as the Rank-Assisted Prefilling (RAP) attack, can effectively extract harmful content from models fine-tuned with the data augmentation defense by selecting low","title":"Matching Ranks Over Probability Yields Truly Deep Safety Alignment","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-23T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2512.05518"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:cc415e426a6cd5caee00ab577f48a6ffcb9a0197957d017866573e49193fcd6597fa7bd066e0ac182a189180cb8c2cbad5d7b22d90302c88b1f5fed54eb79a0e","signer":"crovia.substrate","subject":{"observed_at":"2026-07-23T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2512.05518"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"64dcc7c8eae428d61f66fbe7f7afbb1760d915847dcb93ab21aa782ccc193533","leaf_index":343780,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"13e4669ab6dd74746e4c0248f8aa85ca3bbb52f3b2e0ca045590d1cd8ec9903e","side":"right"},{"sibling":"c221f3027c1c6165b5111c69af4e8e077edbc18ddb09c957ba090089da6e8fbb","side":"right"},{"sibling":"41dcfff50f10a5455533fe4b9658b290ebe0d605a1cc16f43be08a109e5084fc","side":"left"},{"sibling":"3fc80d68be74125e490a9a1ab803b141f981e7fe8e1d9996d8f974bd2a9d03e8","side":"right"},{"sibling":"c423429d1a8c3090b58f8897f311a3b03f9f60ac7fb988ff530039a6c6e7c129","side":"right"},{"sibling":"8c4abba7f73d017263c7cfa7cf00e7dad2248eccd24b930d64f9284ce5a8897d","side":"left"},{"sibling":"84e2bd69f70ff0e814f453c01a295fde8d94bdb3824a1424ab4fd97be4fa4cba","side":"left"},{"sibling":"fbe8bc0f33cd1a297ce2f845b101ed1107ebb67f2c2577c28ed032c1046d3b97","side":"left"},{"sibling":"6ecdcc1e2fb6ab44777fffbb7c8297d723018c63aac9d90c7a1bbebe728d84cf","side":"right"},{"sibling":"93d7d8e0e882d05b2a15bb707a824979a0427907eb47e687c712906674a0d345","side":"left"},{"sibling":"92219a3ef58cd145d94f071b0b9396cec3707812b02a8c7c63f2d0e22340552b","side":"left"},{"sibling":"4eb402d67bd4bf583b0434363061166fe259c34cc6c42adb32dcfbff0a9f5767","side":"left"},{"sibling":"2dd9cb2521044ee7c6b74f2315e0a0253b8df0d04a7b810bbbbe7da5a9788769","side":"left"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"941f71d7ce3990a507b8f485de3f872a51d0e9a8c59405d76c2ae6f4a6af494a","side":"right"},{"sibling":"6281b6f7a93c44e3c4895bc65cfcb6f2be24dd725f4f46540ec022a6e215f4e8","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"6baede22892163664bb2e4d92cf75e6b290761533a6c39491c3afd89bb3a0252","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":343944,"merkle_root":"afab58f71597d393a7a61b857dac2eacb72fd1c04cd1432c2622d7b19309dffd","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260723T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-23T05:38:39Z","sig_algorithm":"ed25519","signature":"1a58fdc6367d0c86f83d2385be748e0800a64bcf9987c92fadca53deda009cd390d2070302c2b6e44841febfd864637c323ba12c7a1e9ae58fdf2a0cf546ac01","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_af132954eb3a9589c6f47dce441cd2ab801f6adebe20b67e834bd5f9f5ec9062"}}