{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_f604c6a474e6348892278e9c9a38795714b2a14b882cb265726f1c76ef1e52ff","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_f604c6a474e6348892278e9c9a38795714b2a14b882cb265726f1c76ef1e52ff","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"d00cbc24eb691d08519f355996d18351b3706bb0a33360a29f269bcb69f693ab","published":"Mon, 13 Jul 2026 00:00:00 -0400","receipt_hash":"d00cbc24eb691d08519f355996d18351b3706bb0a33360a29f269bcb69f693ab","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"d00cbc24eb691d08519f355996d18351b3706bb0a33360a29f269bcb69f693ab","observed_at":"2026-07-13T04:43:08.394955Z","parent_run_hash":"900c1c934245e788564e199a9619f2dc36ec91d9804ddd9c6a40fb42c8a1e1c0","published":"Mon, 13 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.20140v2 Announce Type: replace \nAbstract: Direct Preference Optimization (DPO) is an effective framework for aligning large language models with human preferences, but it struggles with complex reasoning tasks. DPO optimizes for the likelihood of generating preferred over dispreferred responses in their entirety and lacks the granularity to provide feedback on subsections of many-step solutions typical of reasoning tasks. Existing methods excel at either stable preference learning (e.g., DPO variants like KTO and RSO) or structured reasoning (e.g., ReMA's multi-agent RL framework, Tree of Thoughts), but fail to merge these complementary strengths. We propose HiPO (Hierarchical Preference Optimization), an extension of DPO that separates responses into reasoning segments (query clarification and context, reasoning steps, and answer) and computes loss as a weighted sum of the DPO loss for each segment. Our approach enables segment-specific training while maintaining DPO's comp","title":"HiPO: Hierarchical Preference Optimization for Adaptive Reasoning in LLMs","url":"https://arxiv.org/abs/2604.20140","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.20140v2 Announce Type: replace \nAbstract: Direct Preference Optimization (DPO) is an effective framework for aligning large language models with human preferences, but it struggles with complex reasoning tasks. DPO optimizes for the likelihood of generating preferred over dispreferred responses in their entirety and lacks the granularity to provide feedback on subsections of many-step solutions typical of reasoning tasks. Existing methods excel at either stable preference learning (e.g., DPO variants like KTO and RSO) or structured reasoning (e.g., ReMA's multi-agent RL framework, Tree of Thoughts), but fail to merge these complementary strengths. We propose HiPO (Hierarchical Preference Optimization), an extension of DPO that separates responses into reasoning segments (query clarification and context, reasoning steps, and answer) and computes loss as a weighted sum of the DPO loss for each segment. Our approach enables segment-specific training while maintaining DPO's comp","title":"HiPO: Hierarchical Preference Optimization for Adaptive Reasoning in LLMs","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-13T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.20140"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:6bd87869517406ef052e79e37e3562ef70b3632ff934eb23343e2992fd8a1429e26b404bda61ed5adbfc1c52a2801bd1001ec77bd32bb854cdbe9f141e3cb40d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-13T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.20140"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f08cc550c4a51945cc820a15d747a558a469a3b09cedde621ca1a32af6b246bf","leaf_index":309698,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"d3d696ecbb86809403bdd356e4622a7e430479952e9e06cd21d7cdda4a0854ac","side":"right"},{"sibling":"b5fc9b5cad0762b435aaff3980296fe9521e5930df2cc50a22a4a382e83b3962","side":"left"},{"sibling":"dc8a727964b751886305f635e7474f0b66c77eea64c906f2826808ace6b331e5","side":"right"},{"sibling":"92b431ea7e44225410e9a82d8f1122e2453bf3df5031bd9cd4920ac2727df382","side":"right"},{"sibling":"cf2d2286eda9c96ec40bd27fe3d4993e5cbd9bb0671296254b8f361dcc73c7b2","side":"right"},{"sibling":"88c44e323ef86a9c4f17a14b68eb1a15c4b1b49a81438fbb84791655c362f2d7","side":"right"},{"sibling":"42713195b458e33ad755469890443a1ff2740bc55d9158b7553b968ec2d002fc","side":"left"},{"sibling":"49f7764de5dea797b32547b8437f6d371cde9fb33ac6cf784651dcfa196b149a","side":"left"},{"sibling":"6cbb0c4695e74fa8ec17dfde89fa56c1958a1d4dffbbcbe3556da4a69c699ebc","side":"left"},{"sibling":"c4bde3283be97905033d39c1097ac73b82f180de8830384fe6f9f1933f7fa4b9","side":"right"},{"sibling":"3b5f968ebea87e7be458ef7e63c6637988d27ba6d170e1f3794d385cd76ca23f","side":"right"},{"sibling":"5ed534e945b31085c140b50460415e7960b76a4b6266da67d272b853dc94b352","side":"left"},{"sibling":"91010b271bc8eb5253b3549292ef3146e1d85bc9bac7ca36e0862f1204f84e8d","side":"left"},{"sibling":"772fbc112e94d8e574379343387c65503d3cf8fb16ff89090174eccced871a41","side":"left"},{"sibling":"5d50450cae1a230f682b390c8e28ae822ec6c0c04a9bc79b0d27af97306ccccd","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"e9ac4b1e71d3f572751b7984e9ca00893d71628137b4634e82df02cf3db9ab68","side":"right"},{"sibling":"99ff86058e045249bf936a629308be31e4cc71328f0282c4a924f4e6718be5f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":309862,"merkle_root":"f18a76abb66e7cb448986b6541091416ed4ebdde8124f5208a7c7c94bb4165d1","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260713T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-13T05:38:23Z","sig_algorithm":"ed25519","signature":"0488e3527b1d559ba55116220f91e2f6c22bb5358a135e8a9b94a65450b50511702044fa993555662ebca09f005b277f5f6ad0d044ec96a5568e9993c09ef007","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_f604c6a474e6348892278e9c9a38795714b2a14b882cb265726f1c76ef1e52ff"}}