{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_200c0434585d9ddfa064c97dfc2d8c45d756c5e47893d823c0f3f378e5ed45e6","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_200c0434585d9ddfa064c97dfc2d8c45d756c5e47893d823c0f3f378e5ed45e6","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0c0faf95145ae9a9429153d43e32076fdb2410917ea2ca5f7613fbc265e5eb5c","published":"Wed, 10 Jun 2026 00:00:00 -0400","receipt_hash":"0c0faf95145ae9a9429153d43e32076fdb2410917ea2ca5f7613fbc265e5eb5c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0c0faf95145ae9a9429153d43e32076fdb2410917ea2ca5f7613fbc265e5eb5c","observed_at":"2026-06-10T04:43:37.461885Z","parent_run_hash":"23aff1a6f676ba7ca33f70f4ddfae1dd282fb86104d577ce9be510d81a94c5dc","published":"Wed, 10 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2512.06343v3 Announce Type: replace-cross \nAbstract: Reward models are central to Large Language Model (LLM) alignment within the framework of RLHF. The standard objective used in reward modeling is the Bradley-Terry (BT) loss, which learns from pairwise data consisting of chosen and rejected responses. In this work, we analyze the per-sample gradient of BT-loss and show spurious learning signals due to representation distance. In particular, BT gradient norm scales with two distinct components: (1) prediction error, reflected by the difference in predicted rewards between chosen and rejected responses, and critically, (2) representation distance between the pair measured in the output space of the final layer. While the first term captures the intended training signal, the second term can significantly impact the update magnitude and misalign learning. Specifically, pairs with small representation distance often receive vanishingly weak updates, even when misranked, while pairs ","title":"When Distance Distracts: Representation Distance Bias in BT-Loss for Reward Models","url":"https://arxiv.org/abs/2512.06343","vendor":"arxiv_cs_ai"},"summary":"arXiv:2512.06343v3 Announce Type: replace-cross \nAbstract: Reward models are central to Large Language Model (LLM) alignment within the framework of RLHF. The standard objective used in reward modeling is the Bradley-Terry (BT) loss, which learns from pairwise data consisting of chosen and rejected responses. In this work, we analyze the per-sample gradient of BT-loss and show spurious learning signals due to representation distance. In particular, BT gradient norm scales with two distinct components: (1) prediction error, reflected by the difference in predicted rewards between chosen and rejected responses, and critically, (2) representation distance between the pair measured in the output space of the final layer. While the first term captures the intended training signal, the second term can significantly impact the update magnitude and misalign learning. Specifically, pairs with small representation distance often receive vanishingly weak updates, even when misranked, while pairs ","title":"When Distance Distracts: Representation Distance Bias in BT-Loss for Reward Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-10T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2512.06343"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:2a2d6b58cd8bd7416422e3770ba727f90119c3651fea8dbdee4de1b02bc3b17af6f1cd8f3677bd7d25018eb3f1fce787d96f804770fbbf2ce2ff61bfe841ac04","signer":"crovia.substrate","subject":{"observed_at":"2026-06-10T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2512.06343"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"4449eac713cc351b2e201a148804a35d26985882d110ff0c9b96bb54f0100251","leaf_index":226181,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c137cb72d45394897735e3ce828a24387ac874df3f89c8d4c08ae878ea80f631","side":"left"},{"sibling":"5ddffc052aae270c535e58a932f83e9fe781c05ca5c0d08920d8ccf68341615e","side":"right"},{"sibling":"1dfa313fa625ff6edb0a1dc179904e22fa6cf1d3280e1818b8c68f2f13b5a8d3","side":"left"},{"sibling":"95b284510084d918b1ad2a9c7d1b80d3aaa219be7ddf0276e24c2777b5aee5b0","side":"right"},{"sibling":"4a448ca0b556288aa7467569ea351eec178d2b99e8825303ba1af64faa569f4b","side":"right"},{"sibling":"1ca069d6901e9f6a4c5e8b5510457fe734ca64816f6e27648e29f3fcdf8fc719","side":"right"},{"sibling":"a026ba4444dd60937f3a5636d15324f239921221e1299500edf326142aa1039a","side":"right"},{"sibling":"891febe0a62df3512be2c4743bc3c5d3327de005faf591d59848778f0454f464","side":"left"},{"sibling":"b180cfc3f8638c912e90139ec42e2e3fd9b67b3fe342b36e8954314633ae5f61","side":"left"},{"sibling":"a4d17aefe58175050dc159af6246658fcf1c9f3ed57aacf1b350fc3261de4e69","side":"left"},{"sibling":"280b980aa0c7756b0b0cb22658f26466d36f0e70fbc3312cd2311d9898e30b8f","side":"right"},{"sibling":"c98954d4b658b1dda60fe52576fcf9bf21a2d49c67fb63f8c30f16ab5f721938","side":"right"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_200c0434585d9ddfa064c97dfc2d8c45d756c5e47893d823c0f3f378e5ed45e6"}}