{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_edc5de9c3964be952eb727a0753bb4ed9f4e9313d1f068de11a1b5041fdc958c","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_edc5de9c3964be952eb727a0753bb4ed9f4e9313d1f068de11a1b5041fdc958c","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"7f697eca10470d4b98c4f76b4faa2a149fc2acab4175585646a306b9c227ead3","published":"Fri, 29 May 2026 00:00:00 -0400","receipt_hash":"7f697eca10470d4b98c4f76b4faa2a149fc2acab4175585646a306b9c227ead3","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"7f697eca10470d4b98c4f76b4faa2a149fc2acab4175585646a306b9c227ead3","observed_at":"2026-05-29T04:43:58.478092Z","parent_run_hash":"0fcd87efcfe67ccb9952f747541debc16793919a4d20fd71ca0ad5516a0a13ee","published":"Fri, 29 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.29396v1 Announce Type: new \nAbstract: Safety alignment for large language models (LLMs) aims to reduce harmful or unsafe behavior while preserving general utility. However, recent findings reveal that alignment effects can be fragile: lightweight post-alignment manipulations, such as parameter noise, activation noise, or quantization, can easily weaken the intended safety behavior. Prior efforts to improve robustness have primarily focused on data curation, modified alignment objectives, and safety-critical parameter identification, leaving the role of the optimizer itself largely unexplored.\n  In this paper, we are the first to study the robustness of safety alignment from the perspective of the base optimizer. This optimizer-centric view naturally points to zeroth-order optimization, which provides a robustness-oriented signal by evaluating safety alignment under perturbations. Based on this insight, we propose a hybrid framework that first performs standard first-order sa","title":"Aligned but Fragile: Enhancing LLM Safety Robustness via Zeroth-Order Optimization","url":"https://arxiv.org/abs/2605.29396","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.29396v1 Announce Type: new \nAbstract: Safety alignment for large language models (LLMs) aims to reduce harmful or unsafe behavior while preserving general utility. However, recent findings reveal that alignment effects can be fragile: lightweight post-alignment manipulations, such as parameter noise, activation noise, or quantization, can easily weaken the intended safety behavior. Prior efforts to improve robustness have primarily focused on data curation, modified alignment objectives, and safety-critical parameter identification, leaving the role of the optimizer itself largely unexplored.\n  In this paper, we are the first to study the robustness of safety alignment from the perspective of the base optimizer. This optimizer-centric view naturally points to zeroth-order optimization, which provides a robustness-oriented signal by evaluating safety alignment under perturbations. Based on this insight, we propose a hybrid framework that first performs standard first-order sa","title":"Aligned but Fragile: Enhancing LLM Safety Robustness via Zeroth-Order Optimization","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-29T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.29396"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:99e98a16d27cae410301fb9db35fa78de8fcd57001f69561fe4056eab265a1259e2ec42250b54d1a0f49d1cef476ccd729d412ff895746032f25a778f30d050b","signer":"crovia.substrate","subject":{"observed_at":"2026-05-29T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.29396"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"615fbafe8e97ed38dfeac1a1ecf7c93878fdbf822acaa1916ee9b24d1055fd29","leaf_index":157691,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c7e676129445d4389b3e6fe9dba2f98e88b754fc265dedd65703d21d7963ecc8","side":"left"},{"sibling":"8df0cad35693ae5d3bdc944593e925caed06c4ee39d88e173fdeee5faac1dea9","side":"left"},{"sibling":"439b6e2785a8fd29c2148edd4f5f8528cf3e95669adb417a501f435230f76e36","side":"right"},{"sibling":"fe02690f688009f930d561cdba34d269bb6479d664e4bdad9505aca2e5ba2fa4","side":"left"},{"sibling":"bbe317be0b08887cb42c33c89b5f839d03d339c6f80b79d583069268cd5fc435","side":"left"},{"sibling":"1eead04ebe1be8d8dddce54b72650927fcf3982f00346da13e29805b59a5f05a","side":"left"},{"sibling":"f65c4257e1e3e5a8d1432a336a21d8e806501313e0c2888ff215dba80f4c4cfe","side":"left"},{"sibling":"27e5f392a568b11659ecd5aed52502b5e6d88eea7862dffac806a2d7b5a9ef78","side":"left"},{"sibling":"ece7b043912367e6be374d77e09128de032af83a2b87e39b397350dd279ebe83","side":"left"},{"sibling":"39ec45a73732c5ed1abe96972bd3bc32a517083704467dde1c8117578609124d","side":"left"},{"sibling":"68b4895a8015cdde1c1baeaf63a8382cf574ce4d2ad88d596d4217ce2238245e","side":"left"},{"sibling":"0bc831354843fa27f7ba6a4b3080a72fd440b9a76d1bac63429adfbfe6549bca","side":"right"},{"sibling":"995b421824624a8282c7f44e64c64ee35344800f477ae1845b41be14d3fab94c","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"eef0e8906a749d3470f89beeedc723f37a5737010bbb0dcc7cf91515338e5a3e","side":"right"},{"sibling":"1a07e481a9407d71aad078ce854cdeee362163c887fe10f889b0ecf0b5e749ad","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":158251,"merkle_root":"485e6b31fe60c8beba5b394808c7e4c32448b2ff65c2482c480ca0e2a2eda718","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260529T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-29T05:37:37Z","sig_algorithm":"ed25519","signature":"bbf9f005201182fce4f9d94c7a9d01a508b56daf7d9611bd73514f5f616bc059d0e5e1f2edfc95716e6fe08ef5fae38b558cbf7f2fd8f9d5dfe4c34a54c83005","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_edc5de9c3964be952eb727a0753bb4ed9f4e9313d1f068de11a1b5041fdc958c"}}