{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_dcb415f9eafe0e37daaac2ccbea19ba3845f08dec2927ad83473fdbf02ad31ac","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_dcb415f9eafe0e37daaac2ccbea19ba3845f08dec2927ad83473fdbf02ad31ac","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"531fae35ad93110a35cfe83e5f22bff0759deed3eb3eb73cee588936b2e56db8","published":"Fri, 24 Jul 2026 00:00:00 -0400","receipt_hash":"531fae35ad93110a35cfe83e5f22bff0759deed3eb3eb73cee588936b2e56db8","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"531fae35ad93110a35cfe83e5f22bff0759deed3eb3eb73cee588936b2e56db8","observed_at":"2026-07-24T04:43:08.456021Z","parent_run_hash":"b018378f86139a28e6209ec008b31c1282cd1b5c1632dbd43b054c17aa88ab96","published":"Fri, 24 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.20515v1 Announce Type: new \nAbstract: Reinforcement Learning from Human Feedback (RLHF) is critical for aligning Large Language Models (LLMs) with human preferences. However, its efficacy is often compromised by the inherent inconsistency and subjectivity of human annotations. Existing preference optimization frameworks, such as Direct Preference Optimization (DPO), typically treat ambiguous pairs with high annotator disagreement identically to those with unanimous consensus, forcing models to overfit to inconsistent supervision signals and leading to suboptimal alignment. In this work, we propose Reliability-Guided Preference Optimization (RGPO), a robust framework designed to mitigate the impact of inconsistent human feedback. RGPO estimates annotator reliability and infers latent ground truth labels from noisy human feedback to identify robust preferences. Furthermore, we introduce a reliability-aware consistency optimization that dynamically modulates the training object","title":"Reliability-Aware LLM Alignment from Inconsistent Human Feedback","url":"https://arxiv.org/abs/2607.20515","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.20515v1 Announce Type: new \nAbstract: Reinforcement Learning from Human Feedback (RLHF) is critical for aligning Large Language Models (LLMs) with human preferences. However, its efficacy is often compromised by the inherent inconsistency and subjectivity of human annotations. Existing preference optimization frameworks, such as Direct Preference Optimization (DPO), typically treat ambiguous pairs with high annotator disagreement identically to those with unanimous consensus, forcing models to overfit to inconsistent supervision signals and leading to suboptimal alignment. In this work, we propose Reliability-Guided Preference Optimization (RGPO), a robust framework designed to mitigate the impact of inconsistent human feedback. RGPO estimates annotator reliability and infers latent ground truth labels from noisy human feedback to identify robust preferences. Furthermore, we introduce a reliability-aware consistency optimization that dynamically modulates the training object","title":"Reliability-Aware LLM Alignment from Inconsistent Human Feedback","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-24T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.20515"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:f9f366ebb63422820831c5bcf0226721f9b879ad096914a44f622ee0cc1ac259461cb00fdc959256aba89f1327db0a87bdfa8adc01916a41f1f953055d66d803","signer":"crovia.substrate","subject":{"observed_at":"2026-07-24T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.20515"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"b3715455170b29cc96ef3f1d0b475d3419a08ebffabae7d24cd7187eaca2af1a","leaf_index":346988,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"8adafad91b43cb318899e01d31df879d326596771c2717ee3a9b39a8fda4cab8","side":"right"},{"sibling":"54dd0574466cfb9172fbf5ec1fe515e0a4cbaffceca0ba48249ce768fe1b4367","side":"right"},{"sibling":"7773492b9dd9561f5e8df01bb08ea429b78b85e8865eedf09fc5967ac371081c","side":"left"},{"sibling":"723c7649b109b5cb9292354a31b349e5056f72c73102da6592605e50477e3909","side":"left"},{"sibling":"0222a459a14f16bc313fb527b643c2f09817c6cca93ded8a3d93a9f324d80550","side":"right"},{"sibling":"2d8571a85ef8db282687baf52c678be7c1e616f7a8b798c9f0b7c1f94ce5acd7","side":"left"},{"sibling":"6149df3d1da13abe8a81da069f7bd406a0badf8b62cfc2d62ba645f403037400","side":"left"},{"sibling":"ffd3a55995db3d94b2e5c6432ae18b8c5c13fa05a460ebe646d97e4cfd4c196c","side":"right"},{"sibling":"c212fcc83c532c0321804b72fe72b546946bb055d3d482aab19c43f5bebfbf3f","side":"left"},{"sibling":"da189c159d0789c2229cf3731890cc753832dab1aa83e4bbf0fac01955c22cd8","side":"left"},{"sibling":"3a02ed8ea41a09957282e4e27db74ed88cd675156461abb02acb28e2cd257e63","side":"right"},{"sibling":"d3139af8c5ce235438e1c69e4b7afa44ba09129fd86674968434f23e546f423e","side":"left"},{"sibling":"cb89775a838ee16d10fc8da3213420c2012b4d96e8d55cd49939b0887a4b92d3","side":"right"},{"sibling":"252d30ea8052c3bb6b40bc5cc29fc9b9725343d212f84c08fbbae4215a125b00","side":"right"},{"sibling":"f3e45bceed774d2402fa45d41ff5190f295823bd2f216eb90157884150034693","side":"left"},{"sibling":"3cfa2102c0224815c6f3bf73e6710e24103f43f7bf5da1ca2abad1416d9c0890","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"e871fd7edf9b2ad89bce1609a028f5225eea4d14372169bac242420830f86530","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":347413,"merkle_root":"9efe042c5dd6583dfd3b6a58fbfc289807f60bcf2bd2927f10488a54a8ba11fc","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260724T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-24T05:38:42Z","sig_algorithm":"ed25519","signature":"8633c55f558d42994850505218b2862c6134bad2b1c80d4b80736c2fd3ea7a19690ca3498727a8adbdc47791176128a8d64bef0883b888db809f0477355bd00c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_dcb415f9eafe0e37daaac2ccbea19ba3845f08dec2927ad83473fdbf02ad31ac"}}