{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_733ba7bb818adac702ae51cf18fc1a80fe9a991baf08fe208c0623cab8f0af0a","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_733ba7bb818adac702ae51cf18fc1a80fe9a991baf08fe208c0623cab8f0af0a","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"dc042736389dadf1fca4028594d303d32d95ad5d3e4ca3b9f43c8f7e2d4bad61","published":"Tue, 30 Jun 2026 00:00:00 -0400","receipt_hash":"dc042736389dadf1fca4028594d303d32d95ad5d3e4ca3b9f43c8f7e2d4bad61","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"dc042736389dadf1fca4028594d303d32d95ad5d3e4ca3b9f43c8f7e2d4bad61","observed_at":"2026-06-30T04:43:04.087680Z","parent_run_hash":"74f7ab392cc702044101fe24a76a2fdad11164cd79ce725aad6c446a477e89c5","published":"Tue, 30 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.29820v1 Announce Type: cross \nAbstract: In complex continuous-control reinforcement learning tasks, multimodal optimal actions often coincide with uncertain, multimodal return distributions, making reliable value estimation and multimodal exploration challenging. Existing value estimation methods using unimodal Gaussians restrict expressiveness and yield biased estimates. Recent generative policies can represent multimodal actions but often collapse to a few modes and under-explore high-value areas of the action space. Motivated by these challenges, we propose Dual-Flow RL, a unified actor-critic framework that jointly models a continuous return distribution and a multimodal policy distribution using conditional flow matching (CFM). This design supports reliable value estimation and sustained multimodal exploration. To further enhance exploration, we introduce an Entropy-Covariance Exploration Regulator (ECER) that enables state-aware exploration regulation leveraging policy","title":"Dual-Flow Reinforcement Learning with State-Aware Exploration","url":"https://arxiv.org/abs/2606.29820","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.29820v1 Announce Type: cross \nAbstract: In complex continuous-control reinforcement learning tasks, multimodal optimal actions often coincide with uncertain, multimodal return distributions, making reliable value estimation and multimodal exploration challenging. Existing value estimation methods using unimodal Gaussians restrict expressiveness and yield biased estimates. Recent generative policies can represent multimodal actions but often collapse to a few modes and under-explore high-value areas of the action space. Motivated by these challenges, we propose Dual-Flow RL, a unified actor-critic framework that jointly models a continuous return distribution and a multimodal policy distribution using conditional flow matching (CFM). This design supports reliable value estimation and sustained multimodal exploration. To further enhance exploration, we introduce an Entropy-Covariance Exploration Regulator (ECER) that enables state-aware exploration regulation leveraging policy","title":"Dual-Flow Reinforcement Learning with State-Aware Exploration","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-30T04:43:04Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.29820"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:3efde35cee8bd52b97f1f433134df4c6a56846a6bf4840dbef3881b87def71b7a4098e9a6b5a595c99f3137bdfa7b0889b9c934239154e7d43de5b1bac2c6a0d","signer":"crovia.substrate","subject":{"observed_at":"2026-06-30T04:43:04Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.29820"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"bc5f30c3317cd569e70f0a74bc70b998d2e1345241b89c968429b0a08d7b9be4","leaf_index":264950,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"a3393ddfb00334a4abb9ea32ad1228a4c60e8200f0d4d763771c82d484094aba","side":"right"},{"sibling":"f4d00558e74b22eb47f778f89ca6832064abdd06ca7be4100ee785e1420ab20d","side":"left"},{"sibling":"60893f3fad3f341cb8d4071f5b1bc9c8312ad565601a064254c5452f0a21a9e4","side":"left"},{"sibling":"4caf0c7b8932edad1556b87b0191437874cef665c2077ee3dcf7af2c83b2832c","side":"right"},{"sibling":"5eb9b32d02b40649f8b9b1e24ed772e10161505127f9be688aa3af2e4a597e61","side":"left"},{"sibling":"5afa719fed6bf20458cb57f93005e799606deddcec7186b00097a0b05d5361d6","side":"left"},{"sibling":"4330bdf65b90bf9ea57b2b83a3c8e878c009cfc7f28a23b0d78d0b53451c71a6","side":"left"},{"sibling":"86ba07f5dfcbf22e044b5aa0ecc15cd2f29de9218e54da603c4ffe753c522516","side":"left"},{"sibling":"5767f14b55df4212d25ae2f52b8f3d4abff2555c65ba62935ed6cb0a8e617b43","side":"right"},{"sibling":"1549a8883ab3267f958dc2624919e40c65c82958b8005967ed6a4da1247da0ad","side":"left"},{"sibling":"23994bf0974e5c9c7f63a61b4f0a48b0ca756a4adc34a8f85f878e774c37dfbe","side":"right"},{"sibling":"f9b4bed84fa6990c71ad2887c91bda183001f05f6b648f21d1273045a26b11fd","side":"left"},{"sibling":"173d2dc4b29ee04ea41d6d0ebc334c4bc2d46e7ee4230c94765413f24fb4bc42","side":"right"},{"sibling":"112461f7c0ec411116fb5c6c90fe95cea9d8f188b9fe08afe25a837ac02d0071","side":"right"},{"sibling":"ea9488204352c49db8f7daf05eefcd7628ecf9413830346674801a99d0654a94","side":"right"},{"sibling":"6261c13b9922cb657f10d1e5d36ec15d8771cf8766e36c61dcbffb7bed57e396","side":"right"},{"sibling":"fa19aa3faf287618b820bcfceebb366152ad521dd20ef9f51e977816663e448b","side":"right"},{"sibling":"c32f943406b62d1fc59b7f7e243492174c8e1caba8c8a2705f86c773315736e0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":265374,"merkle_root":"9636001ecab173cb6af10dc7c71eb14585daa62f9c0a6f027046f05633156891","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260630T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-30T05:38:03Z","sig_algorithm":"ed25519","signature":"6d4b8fd9b9da856cbb5fba7540877c6a63fa18a5ec3eaf28dc4d6d1c64921c9c0565f8c95f4b7aec0e7744fd7754844f051baf863db708cad768765b416a7b0c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_733ba7bb818adac702ae51cf18fc1a80fe9a991baf08fe208c0623cab8f0af0a"}}