{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_58adb23b7820008b27730000f7a05d71b3d2db5cd30d99d65208d2af3c1ebbf9","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_58adb23b7820008b27730000f7a05d71b3d2db5cd30d99d65208d2af3c1ebbf9","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0f91b16418ae61932b5ee9ae7519b916b1306ea1f1351ebd82bda78b83c34795","published":"Tue, 19 May 2026 00:00:00 -0400","receipt_hash":"0f91b16418ae61932b5ee9ae7519b916b1306ea1f1351ebd82bda78b83c34795","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0f91b16418ae61932b5ee9ae7519b916b1306ea1f1351ebd82bda78b83c34795","observed_at":"2026-05-19T04:43:36.782648Z","parent_run_hash":"fefa4c726316a95c5dda9fc1ca07a38a811cf7ffa9825b2f09b365abacd9b32d","published":"Tue, 19 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2503.02574v2 Announce Type: replace-cross \nAbstract: In this paper, we argue that current safety alignment research efforts for large language models are hindered by many intertwined sources of noise, such as small datasets, methodological inconsistencies, and unreliable evaluation setups. This can, at times, make it impossible to evaluate and compare attacks and defenses fairly, thereby slowing progress. We systematically analyze the LLM safety evaluation pipeline, covering dataset curation, optimization strategies for automated red-teaming, response generation, and response evaluation using LLM judges. At each stage, we identify key issues and highlight their practical impact. We also propose a set of guidelines for reducing noise and bias in evaluations of future attack and defense papers. Lastly, we offer an opposing perspective, highlighting practical reasons for existing limitations. We believe that addressing the outlined problems in future research will improve the field'","title":"LLM-Safety Evaluations Lack Robustness","url":"https://arxiv.org/abs/2503.02574","vendor":"arxiv_cs_ai"},"summary":"arXiv:2503.02574v2 Announce Type: replace-cross \nAbstract: In this paper, we argue that current safety alignment research efforts for large language models are hindered by many intertwined sources of noise, such as small datasets, methodological inconsistencies, and unreliable evaluation setups. This can, at times, make it impossible to evaluate and compare attacks and defenses fairly, thereby slowing progress. We systematically analyze the LLM safety evaluation pipeline, covering dataset curation, optimization strategies for automated red-teaming, response generation, and response evaluation using LLM judges. At each stage, we identify key issues and highlight their practical impact. We also propose a set of guidelines for reducing noise and bias in evaluations of future attack and defense papers. Lastly, we offer an opposing perspective, highlighting practical reasons for existing limitations. We believe that addressing the outlined problems in future research will improve the field'","title":"LLM-Safety Evaluations Lack Robustness","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-19T04:43:36Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2503.02574"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:645a09bd728d3dc793a4d462ed15c46d01774179b154ab73da66fde26361a5d99f66b752481ed6c6b3ebe1e0de5d1283ddbe8fff6079806fb0998f592deabc0b","signer":"crovia.substrate","subject":{"observed_at":"2026-05-19T04:43:36Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2503.02574"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"600a3b01ce814beac4fa51ca634ecedc04282a6311582f47b2d6245501c193c0","leaf_index":143027,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c3a7811761810590a769cb039b0bd4c714ed80e07e805548d5bb4d3aa3f96df4","side":"left"},{"sibling":"a53128dedf451b9429a04bf55825b2cfb1a21d9470d2ef40a741581f352c1435","side":"left"},{"sibling":"4e9c7f91f82d0fa2fb37359c5fcb2ce56a80bb499ffdd62bfb43cfcba676bf2f","side":"right"},{"sibling":"12704e4bcf9ded2f2be95aa5b3ee085493e8ef47e8e78fc602d019deeec55221","side":"right"},{"sibling":"e585c0670879fe6f0ee2c730be67792ee74accb74f58cddb4be57480dc57188e","side":"left"},{"sibling":"3205a2ed9c1def8f0dfe01b2865b2e0d6576e32ca32e78850b604a2feb905af6","side":"left"},{"sibling":"71cfa2f48c45714c85044224a434d4671a31490918bf5f7b7ac68ab93339a056","side":"right"},{"sibling":"a08bb38b15de0c5742c8163821e1b5929df17d7b24ec355bdc5003a27eafc8e7","side":"left"},{"sibling":"78a19962a2f444f055541381645626e3d4e1c8c7c2811bf54eeb89100f89f5b2","side":"right"},{"sibling":"db97141c585f6a1e6bebe92b3ea300ea0f38a2321ca286d85850b11b2dd162a6","side":"left"},{"sibling":"202f1bead178ef3785968d50d3d188264a95192a077654c331612e04a34cbfbe","side":"left"},{"sibling":"72249c8c8b068386e35d16f4bd0bbeb9ba820ca217ef0f0d28396c9fe493f5f0","side":"left"},{"sibling":"ea64599340f7ffdf17ad0cbc1d9401ef8870a347e3847bdc106d06b1673df09c","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"4db1f363729507e27a60851cf6ed334d7b9acdef194ed7d419aba4d2bd367a4a","side":"right"},{"sibling":"a86ee18c45e7fcc408b6007eaece05aa75b2d9ae30252e9e878462b4dffbef7b","side":"right"},{"sibling":"1d18e7663d43ccff0122ecc7ee12645bb16afb607b218e81b1ea2408f863cb78","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":143302,"merkle_root":"999156d40a7c61d9ddd52b7338f3cbda3e68f53bace070c7b616ea194e23b123","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260519T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-19T05:37:30Z","sig_algorithm":"ed25519","signature":"b1a252cc66ff32bed1d10dd88a6b2a200e3856d3dbcfcc4ee55e02e00f3d548e854ed9c544704b222bd5d315492c4a935ba2d90d727c585a67899b0ad602fc05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_58adb23b7820008b27730000f7a05d71b3d2db5cd30d99d65208d2af3c1ebbf9"}}