{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_540b6adebc42a08ee2efc0b677d05fab964eec43c624d960299bfa398bf320ff","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_540b6adebc42a08ee2efc0b677d05fab964eec43c624d960299bfa398bf320ff","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0b004c6e6c1b1c2bf662e4525d8fe97caf82bd51b79fc9e243ae134a083a6686","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"0b004c6e6c1b1c2bf662e4525d8fe97caf82bd51b79fc9e243ae134a083a6686","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0b004c6e6c1b1c2bf662e4525d8fe97caf82bd51b79fc9e243ae134a083a6686","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.18037v2 Announce Type: replace-cross \nAbstract: Reinforcement Learning from Human Feedback (RLHF) or Verifiable Rewards (RLVR) are two key steps in the post-training of modern Language Models (LMs). A common problem is reward hacking, where the policy may exploit inaccuracies of the reward and learn an unintended behavior. Most previous works address this by limiting the policy update with a Kullback-Leibler (KL) penalty towards a reference model. We propose a different framing: Train the LM in a way that biases policy updates towards regions in which the reward is more accurate. First, we derive a theoretical connection between the accuracy of a reward model and the flatness of an optimum at convergence. Gradient regularization (GR) can then be used to bias training to flatter regions and thereby maintain reward model accuracy. We confirm these results by showing that the gradient norm and reward accuracy are empirically correlated in RLHF. We then empirically show that Ref","title":"Gradient Regularization Mitigates Reward Hacking in Reinforcement Learning from Human Feedback and Verifiable Rewards","url":"https://arxiv.org/abs/2602.18037","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.18037v2 Announce Type: replace-cross \nAbstract: Reinforcement Learning from Human Feedback (RLHF) or Verifiable Rewards (RLVR) are two key steps in the post-training of modern Language Models (LMs). A common problem is reward hacking, where the policy may exploit inaccuracies of the reward and learn an unintended behavior. Most previous works address this by limiting the policy update with a Kullback-Leibler (KL) penalty towards a reference model. We propose a different framing: Train the LM in a way that biases policy updates towards regions in which the reward is more accurate. First, we derive a theoretical connection between the accuracy of a reward model and the flatness of an optimum at convergence. Gradient regularization (GR) can then be used to bias training to flatter regions and thereby maintain reward model accuracy. We confirm these results by showing that the gradient norm and reward accuracy are empirically correlated in RLHF. We then empirically show that Ref","title":"Gradient Regularization Mitigates Reward Hacking in Reinforcement Learning from Human Feedback and Verifiable Rewards","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.18037"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4dddeb4f9b4e51e8da5dfe8b40a97add3178289041723fb0f9536f4303aeeaa75314c03792a180cf862b6ea5ebadf34c41f2b4837223eb63c0f8a03c9dbc3b0a","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.18037"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"bae13d393b5b19bae793c15b7cde6e31e65a6bc5a68ee48e5787033d9486dcb8","leaf_index":289365,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"64aa9dc9ef6db42a94f0c884b0c4655995bed64cd4bf051a5d5089702d5ca159","side":"left"},{"sibling":"e20783f61657b4030f1922c19339002ddd1cbdf5387a65893c4db2df309c6c50","side":"right"},{"sibling":"85d19152336c06f4f17b8baa16057937d210080e06dc6960b9e12a02909652c8","side":"left"},{"sibling":"2e8ce0f08fb1df915a254deaae4af32e2811a42021385ea0464c8a3334cf6b28","side":"right"},{"sibling":"4f4e352109564502fba6a4f99f01c58985cb12866f2ddac5c5e2be0371b2ba29","side":"left"},{"sibling":"b4b3fdb042b5e716a678771ca7ab4b0d4901f1c0adfb80037dba93bd66a0d3a6","side":"right"},{"sibling":"267dd112e674315ecd9b3cbdd934dbc9d6f83a3dd743947e87095d624b0d5d16","side":"left"},{"sibling":"5adcd5a480e23093dce11f4b3900b046cf6185bd037ccd9881856faa29f08083","side":"right"},{"sibling":"f6a26c200957df5b056969d2ac473b7e794709bf54454460f447a4af62c6bd58","side":"right"},{"sibling":"c6f2478caecaf381e6b06b04f99195339b88d0db4da957bf2106979ee4a0375c","side":"left"},{"sibling":"19d6dfd29bc47f35fa02e8fe765277ba9cc3e6da5072309f24ebaac5b5f295e3","side":"right"},{"sibling":"8e0ad7889eb2d4b40e5b6c3d8e2eb19d4e202374983f468aa76321823de07a9f","side":"left"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_540b6adebc42a08ee2efc0b677d05fab964eec43c624d960299bfa398bf320ff"}}