{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_10b856331f35a486dc1d3b0654a4e0f89959147b80b1a2bcb153b5d880ebe1ff","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_10b856331f35a486dc1d3b0654a4e0f89959147b80b1a2bcb153b5d880ebe1ff","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a4b60fe45b3b06d545f77992d2f5d32d6b7ae3f99c8fd9e3df570a1cf60580db","published":"Tue, 14 Jul 2026 00:00:00 -0400","receipt_hash":"a4b60fe45b3b06d545f77992d2f5d32d6b7ae3f99c8fd9e3df570a1cf60580db","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a4b60fe45b3b06d545f77992d2f5d32d6b7ae3f99c8fd9e3df570a1cf60580db","observed_at":"2026-07-14T04:43:37.834979Z","parent_run_hash":"66b89520a448b8d9fe7d8f602ef38b82b6c95e57f532ce72de51375b41870477","published":"Tue, 14 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.32027v2 Announce Type: replace-cross \nAbstract: Reward design remains a central bottleneck for autonomous robot policy improvement, especially in long-horizon manipulation tasks where sparse success labels provide too little signal and binary preferences collapse many competing notions of quality into one ambiguous signal. We introduce Freeform Preference Learning (FPL), a method for learning robot policies from freeform human preferences. Rather than asking annotators which of two trajectories is better overall, FPL lets them define natural-language preference axes, such as speed, safety, quality of placement, or carefulness, and provide pairwise preferences along each axis. These annotations are used to learn a language-conditioned reward model that maps a trajectory and preference label to an axis-specific reward. We use this model to train a reward-conditioned policy that optimizes across the multiple human-specified dimensions. Across four real-world and two simulated l","title":"Freeform Preference Learning for Robotic Manipulation","url":"https://arxiv.org/abs/2606.32027","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.32027v2 Announce Type: replace-cross \nAbstract: Reward design remains a central bottleneck for autonomous robot policy improvement, especially in long-horizon manipulation tasks where sparse success labels provide too little signal and binary preferences collapse many competing notions of quality into one ambiguous signal. We introduce Freeform Preference Learning (FPL), a method for learning robot policies from freeform human preferences. Rather than asking annotators which of two trajectories is better overall, FPL lets them define natural-language preference axes, such as speed, safety, quality of placement, or carefulness, and provide pairwise preferences along each axis. These annotations are used to learn a language-conditioned reward model that maps a trajectory and preference label to an axis-specific reward. We use this model to train a reward-conditioned policy that optimizes across the multiple human-specified dimensions. Across four real-world and two simulated l","title":"Freeform Preference Learning for Robotic Manipulation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-14T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.32027"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b589f8baa8ce6cd32d03396e176a0e3e17e154752fe4e802e30f404ec96319f9099e19820ae00cc1d0f2fd7dbe2dbc6b7e8e298546b7f34241b637528d876406","signer":"crovia.substrate","subject":{"observed_at":"2026-07-14T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.32027"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"aa39c19bbd16228622b756e8fda77466144982e7019a7d20e8a635d810a38334","leaf_index":313252,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"94ad46bab1c8054688b9f4160276131fff348c4a042c7b560ea64e4f66b29285","side":"right"},{"sibling":"29d8cbca47f084ae78d802acbbd623f9131fa2c6106163060b4f1329de066a73","side":"right"},{"sibling":"8c89e18ec41d90662a53064a83de0803f84d122aeae5e0a9bdbeff40f6c58c00","side":"left"},{"sibling":"c10145c121dd38308e5bfd443870e53151afcede8c54fa624bfa817580518992","side":"right"},{"sibling":"cbe82ee73584e70d59180755b38d92c655f37278d0d5a24f3791a17eb98b42f2","side":"right"},{"sibling":"278b23f7ac55fd1fdbeec4fc4e51a177cc788fd3e2040cc4da88f88e49324842","side":"left"},{"sibling":"80b26938907f117f51f78fb01183c45c33c68065966037db1d53385866d28084","side":"right"},{"sibling":"986f162bfceab84459f12c93071b6c3d1b7bd7f19c776e546d33962c0b6427f6","side":"left"},{"sibling":"ae8fbfdec46ce02d0a92359c6a034f0439394d13e731da90290f5ca7192e6da4","side":"left"},{"sibling":"8393dc03644000353c5d503b6ec261c022cd2aa85a649223916153bc8af2dc75","side":"left"},{"sibling":"1627ce6908965f2e5fbc7c16bc8e247867478c587014561d0de1359dc2dc0afc","side":"left"},{"sibling":"74e1ba9fd48bdd0f52777dd5b56c10706e73f42629013748eca13ef113cd59db","side":"right"},{"sibling":"682fd39539db5bce908d0ab6f180374728fed6b34a9a8c87c7b4eb5a726e30b8","side":"right"},{"sibling":"184927f71d667ac0c6ebeadb33394626973738b9107c6dc3c0c4949b44acf295","side":"right"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"c4c189d79669979b59d9aa976ee249c54a7e065b4a05bf49463459cc7f37428e","side":"right"},{"sibling":"19475e206bdf2698769a286db2c97c4d3f089741319f9b29a139038c6e511975","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":313381,"merkle_root":"5f7d48580373d0ab0bc6bdf34f26de442f9a86d130946f7ee45addd1a55cb9f4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260714T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-14T05:38:23Z","sig_algorithm":"ed25519","signature":"97364224142ed1b2a8925fed1bfba2e0a8a6a15f8ca0999b934846ad83584a78c7f0f5cd09dbc1c938b35383fcc5c8fa4b723118e30916628879d1c5fd3a9908","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_10b856331f35a486dc1d3b0654a4e0f89959147b80b1a2bcb153b5d880ebe1ff"}}