{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_ecae645ac98be1af170c03287e2f2494c020c1af25b2ec866f48bac5f7b34275","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_ecae645ac98be1af170c03287e2f2494c020c1af25b2ec866f48bac5f7b34275","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6ba885a89390c7779a2a35e59f9f54cbb7942949f711bc5e984ffda3230b740e","published":"Mon, 18 May 2026 00:00:00 -0400","receipt_hash":"6ba885a89390c7779a2a35e59f9f54cbb7942949f711bc5e984ffda3230b740e","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6ba885a89390c7779a2a35e59f9f54cbb7942949f711bc5e984ffda3230b740e","observed_at":"2026-05-18T04:43:11.219741Z","parent_run_hash":"a8aad7414ebb6b75c726f09cd673410576a7f87e191fbdb9ddac99e9b2b95a05","published":"Mon, 18 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.12667v2 Announce Type: replace-cross \nAbstract: The alignment of Large Language Models (LLMs) utilizes Reinforcement Learning from AI Feedback (RLAIF) for non-verifiable domains such as long-form question answering and open-ended instruction following. These domains often rely on LLM based auto-raters to provide granular, multi-tier discrete rewards (e.g., 1-10 rubrics) that are inherently stochastic due to prompt sensitivity and sampling randomness. We empirically verify the stochasticity of auto-raters that can propagate and corrupt standard advantage estimators like GRPO and MaxRL, as a noisy reward samples can skew normalization statistics and degrade the global learning signal. Empirically, sampling more rewards and taking majority voting may reduce the noise and improve performance, but this approach is computationally expensive. To address this bottleneck, we introduce $\\textbf{O}$rdinal $\\textbf{D}$ecomposition for $\\textbf{R}$obust $\\textbf{P}$olicy $\\textbf{O}$ptim","title":"ODRPO: Ordinal Decompositions of Discrete Rewards for Robust Policy Optimization","url":"https://arxiv.org/abs/2605.12667","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.12667v2 Announce Type: replace-cross \nAbstract: The alignment of Large Language Models (LLMs) utilizes Reinforcement Learning from AI Feedback (RLAIF) for non-verifiable domains such as long-form question answering and open-ended instruction following. These domains often rely on LLM based auto-raters to provide granular, multi-tier discrete rewards (e.g., 1-10 rubrics) that are inherently stochastic due to prompt sensitivity and sampling randomness. We empirically verify the stochasticity of auto-raters that can propagate and corrupt standard advantage estimators like GRPO and MaxRL, as a noisy reward samples can skew normalization statistics and degrade the global learning signal. Empirically, sampling more rewards and taking majority voting may reduce the noise and improve performance, but this approach is computationally expensive. To address this bottleneck, we introduce $\\textbf{O}$rdinal $\\textbf{D}$ecomposition for $\\textbf{R}$obust $\\textbf{P}$olicy $\\textbf{O}$ptim","title":"ODRPO: Ordinal Decompositions of Discrete Rewards for Robust Policy Optimization","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-18T04:43:11Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.12667"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:cda96d05162dbc271dc1221013848cd6d6b05fa551866351951e06a41d06f5c59ba99dc2385978dc38f5ab9b48f4bd5844a73bed0c5b83c388b05428f13ec20a","signer":"crovia.substrate","subject":{"observed_at":"2026-05-18T04:43:11Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.12667"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"cb929cba52edf1385ac6acde8d0aa793f5e9a067e554108a13f351d2f08ed4cb","leaf_index":140829,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"9c76ca0b024a5464cf51536803257b80362509555f607ea756949d960b9bfa94","side":"left"},{"sibling":"7cbe50ecd55dac05057bef9a64e0f92d21cc909cb5fdf9e0a85e3e52bf5c231e","side":"right"},{"sibling":"bbbc79ed27c8b88f93c3008e1caa653b56de3df915dfb58c899f94c0e96de510","side":"left"},{"sibling":"d6d61752f75ee3ab22ea04deb01e5850f268f534e86b179c86eebec7c2254453","side":"left"},{"sibling":"f727513692aa02658ee9539f81f537f0867b56b99c709dd467221cf4364e3efa","side":"left"},{"sibling":"2c1eb16147433fe66739128d4f460e8c2a75eee663b7d9dc9c3ad70cfc8c0429","side":"right"},{"sibling":"f5549d405a3735c35bfcb3acd74cae69f4d03370836eec8bbc18e7b3eb016a19","side":"right"},{"sibling":"a87432691ceb26cb93d306a629d23c791a2630efd7fc708d3ad480b3ef31e966","side":"right"},{"sibling":"eec7831167f92e73ff0d6defbc9a917c62b82a9f14cdfbb761db0e7219ab3519","side":"right"},{"sibling":"796add816cf5ab3ce5f338195252cc064685523572fa7fad3982a6e20730395f","side":"left"},{"sibling":"28b78fb112bcf26b6801664db97eb8f52a9bccbf0a7ae6766e11845d443692df","side":"left"},{"sibling":"68d0a4634c1460a19c92edd9480df3aa733b814463e7420d1e14471bf61b2f83","side":"right"},{"sibling":"8af64f275b862349aa3bbb9d5cd7fa9a7fdd5620af3bf1b36b2a4519b0b53bdf","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"b98c2afadb358e5387e88f19588f8343a81b488d9b44a6f7e57a032db3a1b030","side":"right"},{"sibling":"11b0c1591747f09f7c8971a6caa19befcd81317ca9dfd417b143234df4e10c79","side":"right"},{"sibling":"87206f3bcc342797c990d87f7235c01f78d32ca59cfaf8ad18d71afc879ba477","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":140892,"merkle_root":"6cca56ead155990456b8a014cc50bddbe710f409b26e3d1bfa6fb12b0bfcf6bf","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260518T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-18T05:37:30Z","sig_algorithm":"ed25519","signature":"1e1135f7595f79b14fb11f5fa81a2e17ad31b11b44b427a5e40a7d511cd86447daf492babd368ab571cf26404c8c74c450d460130fca4b064eb2760367489a0f","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_ecae645ac98be1af170c03287e2f2494c020c1af25b2ec866f48bac5f7b34275"}}