{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_3b5d1bac6091da692826a12b6098663436fb8f6cae2cdca2ac925a970c25ab85","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_3b5d1bac6091da692826a12b6098663436fb8f6cae2cdca2ac925a970c25ab85","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"dc9f3324687769010bada06dbc2dc47c2bccbffd6e1f45efb3f47d678696b4a5","published":"Tue, 14 Jul 2026 00:00:00 -0400","receipt_hash":"dc9f3324687769010bada06dbc2dc47c2bccbffd6e1f45efb3f47d678696b4a5","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"dc9f3324687769010bada06dbc2dc47c2bccbffd6e1f45efb3f47d678696b4a5","observed_at":"2026-07-14T04:43:37.834979Z","parent_run_hash":"66b89520a448b8d9fe7d8f602ef38b82b6c95e57f532ce72de51375b41870477","published":"Tue, 14 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.10369v1 Announce Type: cross \nAbstract: Flow-matching policies have emerged as an effective policy parameterization for robot learning. They iteratively generate actions from noise, enabling highly expressive modeling of complex and multimodal action distributions. However, prior works observed that scaling these policies with value-gradient reinforcement learning (RL) often leads to training instability. Existing methods attribute this instability to iterative generation and therefore avoid end-to-end value-gradient optimization by sacrificing iterative generation, high expressiveness, or value-gradient optimization. Contrary to prior belief, we show the instability does not stem from iterative generation itself, but from the vanilla sampling strategy originally designed for behavior cloning, which becomes brittle under value-gradient RL. Motivated by this insight, we propose VINE, an RL-oriented sampling method that enables stable end-to-end value-gradient optimization for","title":"VINE: Taming Generative Control Policies for Reinforcement Learning","url":"https://arxiv.org/abs/2607.10369","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.10369v1 Announce Type: cross \nAbstract: Flow-matching policies have emerged as an effective policy parameterization for robot learning. They iteratively generate actions from noise, enabling highly expressive modeling of complex and multimodal action distributions. However, prior works observed that scaling these policies with value-gradient reinforcement learning (RL) often leads to training instability. Existing methods attribute this instability to iterative generation and therefore avoid end-to-end value-gradient optimization by sacrificing iterative generation, high expressiveness, or value-gradient optimization. Contrary to prior belief, we show the instability does not stem from iterative generation itself, but from the vanilla sampling strategy originally designed for behavior cloning, which becomes brittle under value-gradient RL. Motivated by this insight, we propose VINE, an RL-oriented sampling method that enables stable end-to-end value-gradient optimization for","title":"VINE: Taming Generative Control Policies for Reinforcement Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-14T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.10369"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:c43bd779c57566eafb5dab347d395d9de454e764db4aa50af8c71bd3b369c736da0b87f6d077b3d5087592f6a982fefa094fbd85187e8aec1465d77c0ab6ef05","signer":"crovia.substrate","subject":{"observed_at":"2026-07-14T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.10369"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8f7ae9740e75d4ee28108cc90a689455e1c62f47f62b183b63409cf56fdaad46","leaf_index":312982,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"d790fba746d001972a629c32d9870a61365eb522cd865a0103eeb3b624562cc9","side":"right"},{"sibling":"18d9441a68d4fbb8fe084ad8707414a0e9b37d980b5e201aa49cba35027320b3","side":"left"},{"sibling":"68036d147814c692f32ddece0a1c98ac12578a3c9722b4058924b034ed7b7e39","side":"left"},{"sibling":"03676b04806d6fe2cd102d651b79e98ee0c9f80a3950f508230b9fcaf5330151","side":"right"},{"sibling":"b831162626905f6403eaa0af66d34653ad0bc160298a50f379f950201e91b230","side":"left"},{"sibling":"5758228ed940f99f209c19fad6ec953e1ddc1c4589b769cea9254b3555b44184","side":"right"},{"sibling":"77158939eae924bcec6e3a79adbe847f8d8264ba44fc508491982dfcda972764","side":"right"},{"sibling":"018f1c17268dcf6ba6107e8d52410ad70a49124aa289c7dcb4ed6c05713b1cf0","side":"left"},{"sibling":"ca17e703a3fca19bfbb090cbed9e2450594a9c237665dd9a566195d3ef3072f7","side":"right"},{"sibling":"8393dc03644000353c5d503b6ec261c022cd2aa85a649223916153bc8af2dc75","side":"left"},{"sibling":"1627ce6908965f2e5fbc7c16bc8e247867478c587014561d0de1359dc2dc0afc","side":"left"},{"sibling":"74e1ba9fd48bdd0f52777dd5b56c10706e73f42629013748eca13ef113cd59db","side":"right"},{"sibling":"682fd39539db5bce908d0ab6f180374728fed6b34a9a8c87c7b4eb5a726e30b8","side":"right"},{"sibling":"184927f71d667ac0c6ebeadb33394626973738b9107c6dc3c0c4949b44acf295","side":"right"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"c4c189d79669979b59d9aa976ee249c54a7e065b4a05bf49463459cc7f37428e","side":"right"},{"sibling":"19475e206bdf2698769a286db2c97c4d3f089741319f9b29a139038c6e511975","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":313381,"merkle_root":"5f7d48580373d0ab0bc6bdf34f26de442f9a86d130946f7ee45addd1a55cb9f4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260714T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-14T05:38:23Z","sig_algorithm":"ed25519","signature":"97364224142ed1b2a8925fed1bfba2e0a8a6a15f8ca0999b934846ad83584a78c7f0f5cd09dbc1c938b35383fcc5c8fa4b723118e30916628879d1c5fd3a9908","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_3b5d1bac6091da692826a12b6098663436fb8f6cae2cdca2ac925a970c25ab85"}}