{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c42a7bd8a7ee83c561cbea221275ab8b76a6e7d12343584c3263f7bb19e4aaea","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c42a7bd8a7ee83c561cbea221275ab8b76a6e7d12343584c3263f7bb19e4aaea","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"acc46469696b806f12c44ef52b16cf77215d56878d918ba20613f5edce5be46d","published":"Tue, 16 Jun 2026 00:00:00 -0400","receipt_hash":"acc46469696b806f12c44ef52b16cf77215d56878d918ba20613f5edce5be46d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"acc46469696b806f12c44ef52b16cf77215d56878d918ba20613f5edce5be46d","observed_at":"2026-06-16T04:43:43.281320Z","parent_run_hash":"eb6edcf82c3507c59161a4ab46d2e904e507004f44677402bb24d106997ed7c2","published":"Tue, 16 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.14801v1 Announce Type: cross \nAbstract: Flow-matching and diffusion policies are expressive action generators, but optimizing them with temporal-difference reinforcement learning (RL) remains difficult. Effective policy extraction requires exploiting the critic's action gradient, yet directly backpropagating this signal through a multi-step denoising process can be numerically unstable. Existing methods work around this either by discarding gradient information, distilling the policy into a simpler one-step actor, or repeatedly fine-tuning the denoising policy as the critic improves. We propose QPILOTS, a method that leaves the original policy unmodified and steers the denoising process at inference time. At each denoising step, instead of evaluating the critic on the noisy intermediate action where critic predictions are unreliable, we first project that intermediate state to an estimate of the final clean action and compute the critic gradient there. We introduce two varia","title":"QPILOTS: Efficient Test-Time Q-Steering for Flow Policies","url":"https://arxiv.org/abs/2606.14801","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.14801v1 Announce Type: cross \nAbstract: Flow-matching and diffusion policies are expressive action generators, but optimizing them with temporal-difference reinforcement learning (RL) remains difficult. Effective policy extraction requires exploiting the critic's action gradient, yet directly backpropagating this signal through a multi-step denoising process can be numerically unstable. Existing methods work around this either by discarding gradient information, distilling the policy into a simpler one-step actor, or repeatedly fine-tuning the denoising policy as the critic improves. We propose QPILOTS, a method that leaves the original policy unmodified and steers the denoising process at inference time. At each denoising step, instead of evaluating the critic on the noisy intermediate action where critic predictions are unreliable, we first project that intermediate state to an estimate of the final clean action and compute the critic gradient there. We introduce two varia","title":"QPILOTS: Efficient Test-Time Q-Steering for Flow Policies","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-16T04:43:43Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.14801"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:3a25ec5c379af08d9e7d42c8e22a5b5bcad972d218ef78e829b998332402c6ed65ed253b159351631dda7754f5813f05ddc23d7fdbf81e25fac8fec62f503f02","signer":"crovia.substrate","subject":{"observed_at":"2026-06-16T04:43:43Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.14801"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1d3fbd5953077a7f08753f4fd5a0104c26f655d13ac363b237e5fbcaf14d3841","leaf_index":230344,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"892b6e00acadc07eb06652d690d8d00ce2435c1aa7a0dac7af65274ff6ec3bbb","side":"right"},{"sibling":"19f5ea98372fd5d7fd8b24c1cb87e2b1d21164f955cdaa00a87ff50c6832b807","side":"right"},{"sibling":"91052e9c2b059dff485177f73098ae97c64338a0950d4433e495749345a1607e","side":"right"},{"sibling":"06bd16ec1383d4635ae0c65b4c4ba2a60368f5808624f64f15c7779a25960cf4","side":"left"},{"sibling":"318546be5ccf8bd8790ab20c2b643e00956613909d28c1356b00fbaa8092afcb","side":"right"},{"sibling":"e3c605ba32c40253fcf54fdc89d84473f8af121e30ad808ced2b295d5043cb8d","side":"right"},{"sibling":"5aa53f259ebf93fa9ac6b1472f46a3e7bcb96175ba6f54b9c2dc9f5fd9ef8711","side":"left"},{"sibling":"a920a6db38b29333b686b08bcb39326c5a67bb5f4588a0647130fe767a058e04","side":"left"},{"sibling":"03ebe4791d56166247160f6ee7788c15745bdd7e274c62ec21b2438669941587","side":"left"},{"sibling":"74897e850164dddc689c3c65b33f9bae0268ab0bf429867a4e193d9b9b685040","side":"left"},{"sibling":"bde25d7e94e64717e426a97f6fcb4907e92b5c61fc89d92d7e0947a2249c3f6b","side":"right"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c42a7bd8a7ee83c561cbea221275ab8b76a6e7d12343584c3263f7bb19e4aaea"}}