{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_fccbb2d13ada8ee182ffe490fade588df3ad4a8b02245875a853c3cac10a5fc6","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_fccbb2d13ada8ee182ffe490fade588df3ad4a8b02245875a853c3cac10a5fc6","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"4330a3b928469f2d7c5a283d2cb8e960ab2eb01c16f1001eb51aa2c288765550","published":"Thu, 23 Jul 2026 00:00:00 -0400","receipt_hash":"4330a3b928469f2d7c5a283d2cb8e960ab2eb01c16f1001eb51aa2c288765550","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"4330a3b928469f2d7c5a283d2cb8e960ab2eb01c16f1001eb51aa2c288765550","observed_at":"2026-07-23T04:43:18.446221Z","parent_run_hash":"b3f5e4095688e31d15c25de2607bca42111343bdfdac967d48e47905415ddea0","published":"Thu, 23 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.19824v1 Announce Type: new \nAbstract: LLM preference alignment aims to optimize models toward human preferences across diverse user instructions. Reinforcement learning has become a major post-training approach for this goal, but existing proxy rewards are often outcome-level, mainly evaluating the final response while providing limited guidance for the reasoning trajectory. This can make credit assignment coarse when multiple responses receive similar final scores, leaving trajectory-level preferences under-specified. To address this limitation, we propose Thinking Checklist Reward (TCR), a process-oriented reward for RL-based preference alignment. TCR converts preference pairs into sample-specific thinking checklists and uses them to evaluate whether the generated reasoning trace addresses the preference-implied considerations. To reduce overlap with outcome-level supervision, TCR further introduces an exponential moving average (EMA) residual formulation to isolate a comp","title":"Rewarding Better Thinking for LLM Preference Alignment","url":"https://arxiv.org/abs/2607.19824","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.19824v1 Announce Type: new \nAbstract: LLM preference alignment aims to optimize models toward human preferences across diverse user instructions. Reinforcement learning has become a major post-training approach for this goal, but existing proxy rewards are often outcome-level, mainly evaluating the final response while providing limited guidance for the reasoning trajectory. This can make credit assignment coarse when multiple responses receive similar final scores, leaving trajectory-level preferences under-specified. To address this limitation, we propose Thinking Checklist Reward (TCR), a process-oriented reward for RL-based preference alignment. TCR converts preference pairs into sample-specific thinking checklists and uses them to evaluate whether the generated reasoning trace addresses the preference-implied considerations. To reduce overlap with outcome-level supervision, TCR further introduces an exponential moving average (EMA) residual formulation to isolate a comp","title":"Rewarding Better Thinking for LLM Preference Alignment","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-23T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.19824"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b69005ec6ecdeae31171999ebf6f28942a6e23e2779f15bd2dc346fe0f717faf0e6cde96749f938e24855b65e6c4952bf6d7eaa0d263cec7e01f31bdb4883605","signer":"crovia.substrate","subject":{"observed_at":"2026-07-23T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.19824"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"661795182a19ab589bc2d1d7f995e2450109235e919a30280efc0ec331a0f0a8","leaf_index":343620,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"3ec70377d108667d965ad302abee48af44cefb8be72e991b8356c08675ee5e76","side":"right"},{"sibling":"055c9e4d1a8f0c696c1e24619bf0dd2d6ba6ac117f22dfb60f44a4f753f2f383","side":"right"},{"sibling":"c6dd4cb34ed9a1418bc5e4606ee242ca06dd81f3f0cd06b6bf4c42d870fbc7ed","side":"left"},{"sibling":"65f5e74b364e1c7b8c55929d0a8a5933a915592cb6d5f6f9f2d3c1a46241f79c","side":"right"},{"sibling":"26d2880265f6da48218f56905221265bb8b2535b965dd327d3c06b4440dcf894","side":"right"},{"sibling":"95b4bb856076330cc6f380bc7a4d635f0572d9bf95e944aad2740b8c3a54627a","side":"right"},{"sibling":"f1bdad0b747ea102cc58af1ef43ba9823c47e1e8da7945feed779a074a032485","side":"left"},{"sibling":"51ebf5c79aac8726e953f8d1f4d9fb442a2733a55a74c3ce7be5e84fcff3f592","side":"right"},{"sibling":"6ecdcc1e2fb6ab44777fffbb7c8297d723018c63aac9d90c7a1bbebe728d84cf","side":"right"},{"sibling":"93d7d8e0e882d05b2a15bb707a824979a0427907eb47e687c712906674a0d345","side":"left"},{"sibling":"92219a3ef58cd145d94f071b0b9396cec3707812b02a8c7c63f2d0e22340552b","side":"left"},{"sibling":"4eb402d67bd4bf583b0434363061166fe259c34cc6c42adb32dcfbff0a9f5767","side":"left"},{"sibling":"2dd9cb2521044ee7c6b74f2315e0a0253b8df0d04a7b810bbbbe7da5a9788769","side":"left"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"941f71d7ce3990a507b8f485de3f872a51d0e9a8c59405d76c2ae6f4a6af494a","side":"right"},{"sibling":"6281b6f7a93c44e3c4895bc65cfcb6f2be24dd725f4f46540ec022a6e215f4e8","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"6baede22892163664bb2e4d92cf75e6b290761533a6c39491c3afd89bb3a0252","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":343944,"merkle_root":"afab58f71597d393a7a61b857dac2eacb72fd1c04cd1432c2622d7b19309dffd","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260723T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-23T05:38:39Z","sig_algorithm":"ed25519","signature":"1a58fdc6367d0c86f83d2385be748e0800a64bcf9987c92fadca53deda009cd390d2070302c2b6e44841febfd864637c323ba12c7a1e9ae58fdf2a0cf546ac01","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_fccbb2d13ada8ee182ffe490fade588df3ad4a8b02245875a853c3cac10a5fc6"}}