{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_61483bec7bd526bb44ee3ad5376e8124ff37817db3da6395711352cfb99fb736","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_61483bec7bd526bb44ee3ad5376e8124ff37817db3da6395711352cfb99fb736","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"5038ddaa1f3d5fec49ef792ae545ff7f5d2785a60d17d6e3711fdc93163ae638","published":"Wed, 27 May 2026 00:00:00 -0400","receipt_hash":"5038ddaa1f3d5fec49ef792ae545ff7f5d2785a60d17d6e3711fdc93163ae638","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"5038ddaa1f3d5fec49ef792ae545ff7f5d2785a60d17d6e3711fdc93163ae638","observed_at":"2026-05-27T04:43:18.926230Z","parent_run_hash":"6f581915edab4326e2b95fed7c82c2ee149e978d6d7c2443439442a927c31dfa","published":"Wed, 27 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.27355v1 Announce Type: new \nAbstract: Reinforcement Learning from Human Feedback (RLHF) is the standard method to align Large Language Models (LLMs) with human preferences. In this work, we introduce alignment tampering, a potential vulnerability where the LLM undergoing alignment influences the preference dataset, causing RLHF to amplify undesired behaviors. This arises from core limitations of RLHF: (1) preference datasets are constructed from the LLM's own outputs, allowing it to influence them, and (2) pairwise comparisons only indicate which response is better, not why. These limitations can be exploited to cause alignment tampering. For example, if an LLM generates biased responses with higher quality, annotators will prefer them based on quality. However, preference labels do not distinguish quality from bias, and the reward model inherits this limitation. Optimizing such rewards through reinforcement learning or best-of-N sampling can amplify misaligned biases. Our e","title":"Alignment Tampering: How Reinforcement Learning from Human Feedback Is Exploited to Optimize Misaligned Biases","url":"https://arxiv.org/abs/2605.27355","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.27355v1 Announce Type: new \nAbstract: Reinforcement Learning from Human Feedback (RLHF) is the standard method to align Large Language Models (LLMs) with human preferences. In this work, we introduce alignment tampering, a potential vulnerability where the LLM undergoing alignment influences the preference dataset, causing RLHF to amplify undesired behaviors. This arises from core limitations of RLHF: (1) preference datasets are constructed from the LLM's own outputs, allowing it to influence them, and (2) pairwise comparisons only indicate which response is better, not why. These limitations can be exploited to cause alignment tampering. For example, if an LLM generates biased responses with higher quality, annotators will prefer them based on quality. However, preference labels do not distinguish quality from bias, and the reward model inherits this limitation. Optimizing such rewards through reinforcement learning or best-of-N sampling can amplify misaligned biases. Our e","title":"Alignment Tampering: How Reinforcement Learning from Human Feedback Is Exploited to Optimize Misaligned Biases","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-27T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.27355"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:474a5aa2ad14ab6d8d12ea2515f4c9a235cd5d6588b346a329683ab39ec0b226c230aaf93b981de893e94df2ccf8690aeaddacf37062963700f8252990af7504","signer":"crovia.substrate","subject":{"observed_at":"2026-05-27T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.27355"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1581d2beeaeaaab43855b4faf2e3507e2646688c943c67cddb8d493752c7f13f","leaf_index":153620,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"b4c4b34e4d84e4ffd6cb49f63b42023e712e88cb495aa2eae8be2db233e66a66","side":"right"},{"sibling":"716ffae788d44cf2664ba75f01e6d5ad6618f60981e6b62362a74be0ab5db4e1","side":"right"},{"sibling":"d077074f799a1f3895c08e8ae457fe9237f5da3c0876863fc1dca70ac7b7dce1","side":"left"},{"sibling":"c12eca5ada4613884b86bbf1d8e4801d915a891b79945b4f1cd5c24d45a4953c","side":"right"},{"sibling":"e371da8a53a967c4ec17188e62b641899c98217eb6ff6d655b999ce60c90b7c5","side":"left"},{"sibling":"486252b79fe1553ac34a26e597a76a2c2fce3b8e14ea52e5d3552d836c6a8626","side":"right"},{"sibling":"20fef87d78d7a367f04b7046503b17acf0a805917fa54a478472e2f7833e4d1b","side":"right"},{"sibling":"e6fcc8fd02ff764f6e74465d070f84e313e858647d901e3ee00b83931106f26c","side":"right"},{"sibling":"c56e46802f42ead6bcf79619c80428b6dc22e9afc9611a87f0d04a5ef9acadd7","side":"right"},{"sibling":"3165125427a29042fc9d02858a59a68858dbdcb2e19d1afa5f6dd6a95cfbce6a","side":"right"},{"sibling":"04b9a68b8ec6fa37251564383c685c23ce69e5e031df4eae69f79a3a334b68bf","side":"right"},{"sibling":"816f233274bb10f5a122aac086a0c8c697b78fec67a4af55190bb596b7506fab","side":"left"},{"sibling":"d415e6939aee710631f5062799379b547d2c3e3d9a68f263bbb5a693285ab2ca","side":"left"},{"sibling":"374c02d15fb12bd356c179c94766043a982052c6132af8bfc15361b431ffa9f7","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"990705096edf483cc877217308f731dc42d6f6d99f82880167bbdbbfef32560a","side":"right"},{"sibling":"dd265753d95fa2e2fb4f5768e37fab6f691ccff09ad60d0910ff7dc23bac9226","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":154065,"merkle_root":"4993cfdc172e7880b60667f16789dc2e831ff000f81bb1ecba248e73f1510eca","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260527T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-27T05:37:36Z","sig_algorithm":"ed25519","signature":"76ecf118011540405e96506e6219752df04a2850632f2903dc6f10e08b98bc5a42c8d9e1cb5b7c5bf714479806a403df5f34399afa40c23fbb71493a1f77bd0c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_61483bec7bd526bb44ee3ad5376e8124ff37817db3da6395711352cfb99fb736"}}