{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_6948993507976da8236bd48c46ff5146bad0ad87e72ce36797f98912d093416d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_6948993507976da8236bd48c46ff5146bad0ad87e72ce36797f98912d093416d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a1052ac907ff5ed2e631e934e84b7d7c66a92f967c935fe9f6cc4925ac0c306c","published":"Wed, 20 May 2026 00:00:00 -0400","receipt_hash":"a1052ac907ff5ed2e631e934e84b7d7c66a92f967c935fe9f6cc4925ac0c306c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a1052ac907ff5ed2e631e934e84b7d7c66a92f967c935fe9f6cc4925ac0c306c","observed_at":"2026-05-20T04:43:44.562035Z","parent_run_hash":"5f904c2c2fecb6b44f2adce8bdc9de914b9a39f7c4fdd7bf086a6c2361a30f8c","published":"Wed, 20 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2510.18924v3 Announce Type: replace-cross \nAbstract: Reinforcement learning from human feedback (RLHF) or verifiable rewards (RLVR), the standard paradigm for aligning LLMs or building recent SOTA reasoning models, is highly sensitive to noise from inconsistent or erroneous rewards. Yet, the interaction between such noise and widely used group-based policy optimization methods remains underexplored. We introduce a noise-robust Group Relative Policy Optimization (GRPO) and Done Right GRPO (Dr.GRPO) framework that explicitly models reward corruption as Bernoulli noise. Our method applies noise correction after estimating reward flip probabilities to debias the learning signal, yielding provably unbiased gradient estimates. Theoretical analysis shows that group-based methods inherently mitigate individual-level noise, and our correction strategy amplifies this robustness. Empirically, we observe consistent improvements across math and code tasks when applying our noise correction to","title":"Noise-corrected GRPO: From Noisy Rewards to Unbiased Gradients","url":"https://arxiv.org/abs/2510.18924","vendor":"arxiv_cs_ai"},"summary":"arXiv:2510.18924v3 Announce Type: replace-cross \nAbstract: Reinforcement learning from human feedback (RLHF) or verifiable rewards (RLVR), the standard paradigm for aligning LLMs or building recent SOTA reasoning models, is highly sensitive to noise from inconsistent or erroneous rewards. Yet, the interaction between such noise and widely used group-based policy optimization methods remains underexplored. We introduce a noise-robust Group Relative Policy Optimization (GRPO) and Done Right GRPO (Dr.GRPO) framework that explicitly models reward corruption as Bernoulli noise. Our method applies noise correction after estimating reward flip probabilities to debias the learning signal, yielding provably unbiased gradient estimates. Theoretical analysis shows that group-based methods inherently mitigate individual-level noise, and our correction strategy amplifies this robustness. Empirically, we observe consistent improvements across math and code tasks when applying our noise correction to","title":"Noise-corrected GRPO: From Noisy Rewards to Unbiased Gradients","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-20T04:43:44Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2510.18924"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:d3313d1dc4cf08c2b17621cb6f81a7fca2189021e84aa9c57b53a19543f7a486a3cf9c84de6ad1c3f015d1825f46d73739a0fa096fee8eb9b0298f05dad41a06","signer":"crovia.substrate","subject":{"observed_at":"2026-05-20T04:43:44Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2510.18924"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"3cc5449fd2a9b1efeb336175ed7934cdbf2b6dce2dc6b105d38d1dcd63f6d5aa","leaf_index":145268,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"9f2a3364cb9c5e212031aecd1077f10eacd4202bed37d71070ab249534035316","side":"right"},{"sibling":"1869d9f70a09d7d69483311dd54368a0a49600f16fb3d0cc2feef349b04e638a","side":"right"},{"sibling":"1e4ba3b3feff67ea8a37f64991e0e56d879af68ed735be3088fd4e45ee059bb0","side":"left"},{"sibling":"bd49f926c992e7278c8aba2feaa9e16b6e7d6d7294f77a245f2004bbf0b3fd6a","side":"right"},{"sibling":"2d39257204060484eb4366873a43f6b2fe4aaacf5433657f39607d37de515cd7","side":"left"},{"sibling":"b891f8102dbec11b5872d47b2e640beca78f5f15b3e6d01c8d7bbebe1e3b209f","side":"left"},{"sibling":"8abfb0c0cfeb30fba16eb01548a8b7ce35f077e44d727cc0fe4a62d471800c6e","side":"left"},{"sibling":"79fb6d27e8a49dd15a97fca96f52d62ff5ee7213d44a5526459d33f42a050928","side":"right"},{"sibling":"f073aa7be27ee7e1d9eb0de5f129f7e9bae0c584fb959378c395b65831dc1d1d","side":"left"},{"sibling":"1526885f19d1fadf6955cf519dbc4e62d593a4bba99d741e8c301740a7068233","side":"left"},{"sibling":"e3a7d5c07f161682d61bd453ffc02ecdf87cfeda70f986d6650017f9d2d6b265","side":"left"},{"sibling":"edbc49f08e5b92291934c05c9e6efd270a6b0698d8d2fa474006366027dfe098","side":"right"},{"sibling":"3e4df6e7457cecbf36f350375e72dcab336a3984422e4c406ef809e4e2944e96","side":"left"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"be08fedc4e72a6fb56606f66812fae7317e09690b9acb18385f4ab117a981238","side":"right"},{"sibling":"0534329a7475dc9df51998c83c16892126679dade0fa34182f21e869599386c7","side":"right"},{"sibling":"2d24720928ead0e7670650eb55f558c4f20e4c18df376f47ba72cfa8cf0ed344","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":147301,"merkle_root":"08903d7159c3b38eeeeafc09eab15139ea417f1d94f02f1fbc87296b37db840a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260521T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-21T18:37:33Z","sig_algorithm":"ed25519","signature":"905f2924632dfa2970c8690285f5b5d4a1d891d0e0ef1cbc404ebec2fd937215ac768e16f0a9f28b18977a55ae0bfd226db6833ae7ef588729054117d2da7303","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_6948993507976da8236bd48c46ff5146bad0ad87e72ce36797f98912d093416d"}}