{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_16fe981b51950475c956e08f9938305d81b267b9a27a8753ded7bbcd8cacee41","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_16fe981b51950475c956e08f9938305d81b267b9a27a8753ded7bbcd8cacee41","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"111f5cb8519d98e8dee58495e5911b889f716af13bfcea1201c5ff6f709082ff","published":"Fri, 15 May 2026 00:00:00 -0400","receipt_hash":"111f5cb8519d98e8dee58495e5911b889f716af13bfcea1201c5ff6f709082ff","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"111f5cb8519d98e8dee58495e5911b889f716af13bfcea1201c5ff6f709082ff","observed_at":"2026-05-15T04:43:17.611638Z","parent_run_hash":"5c64f85625fabd323e9c4a1cf068c012fb88a248deda9a9ac702fb2f9799f2e5","published":"Fri, 15 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.14297v1 Announce Type: cross \nAbstract: We study reinforcement learning in hybrid discrete-continuous action spaces, such as settings where the discrete component selects a regime (or index) and the continuous component optimizes within it -- a structure common in robotics, control, and operations problems. Standard model-free policy gradient methods rely on score-function (SF) estimators and suffer from severe credit-assignment issues in high-dimensional settings, leading to poor gradient quality. On the other hand, differentiable simulation largely sidesteps these issues by backpropagating through a simulator, but the presence of discrete actions or non-smooth dynamics yields biased or uninformative gradients. To address this, we propose Hybrid Policy Optimization (HPO), which backpropagates through the simulator wherever smoothness permits, using a mixed gradient estimator that combines pathwise and SF gradients while maintaining unbiasedness. We also show how problems wi","title":"Policy Optimization in Hybrid Discrete-Continuous Action Spaces via Mixed Gradients","url":"https://arxiv.org/abs/2605.14297","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.14297v1 Announce Type: cross \nAbstract: We study reinforcement learning in hybrid discrete-continuous action spaces, such as settings where the discrete component selects a regime (or index) and the continuous component optimizes within it -- a structure common in robotics, control, and operations problems. Standard model-free policy gradient methods rely on score-function (SF) estimators and suffer from severe credit-assignment issues in high-dimensional settings, leading to poor gradient quality. On the other hand, differentiable simulation largely sidesteps these issues by backpropagating through a simulator, but the presence of discrete actions or non-smooth dynamics yields biased or uninformative gradients. To address this, we propose Hybrid Policy Optimization (HPO), which backpropagates through the simulator wherever smoothness permits, using a mixed gradient estimator that combines pathwise and SF gradients while maintaining unbiasedness. We also show how problems wi","title":"Policy Optimization in Hybrid Discrete-Continuous Action Spaces via Mixed Gradients","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-15T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.14297"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:e50593cee700d1fe0e77c292b0e601d49078d441caf878a52b8d0af855c3e3208d3399373dd43bf16ee04fd3b6a6e3d569167d874f774a9b4a22dbeea048bc00","signer":"crovia.substrate","subject":{"observed_at":"2026-05-15T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.14297"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"ac9d7ee8308e77d3852c6eb116a099ae46a84b1ca40e7233a71d93d5b3b6ee78","leaf_index":134557,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"9fa5f51996c5f3661363d924db529b986e2c776b47eddb558c5edca4b3b3b132","side":"left"},{"sibling":"49ed2b2987b622d11d752e57cfe59a164ccec4a5601035fc9e96b008bb3a8943","side":"right"},{"sibling":"6543697e40464fa7c725c3eb30e47055ed1193fa7b3b37007a6b2107f48ccce4","side":"left"},{"sibling":"81a7c95ad344599691054501e1e812a4fa59967288f29a02bec0f37ca582465b","side":"left"},{"sibling":"9239c32c11757fadb0083b256ebec76318f286749ad3ba11bd659a5ebe4da1db","side":"left"},{"sibling":"6e1d6260a3d850b194e2a375f8310684b0e177404f2bcdaa038ff5c91a39748b","side":"right"},{"sibling":"1fcc844afba4d276ea40006549a4217e03b90b0c4aa39feb95e5883e34363df9","side":"right"},{"sibling":"cbbed0225164fe227925f0fc2929e9ecbc520f5a86b19035c412206d51935e10","side":"left"},{"sibling":"4711a7f4329f1874c3aa1ae93c336e1d4fa402bdd4b3d766c2a95304ea226882","side":"left"},{"sibling":"8ce2a4687a8ceb409df2e1cb10e44a610b21dc294e518a8550b6d1122335ca2b","side":"right"},{"sibling":"36672459e5ed50c64ee1842b69cb6d2eb682c2a04844555be8d124257571994a","side":"left"},{"sibling":"727783827652adfa99c455bd80a01bfb33836228e51068b4f654ef3da468ca69","side":"left"},{"sibling":"623194cd30880ed223e306737fdb111aa0d781751bfc47553c404a6af6aad2c4","side":"right"},{"sibling":"fc8f53ed42756907fb79ee19a4ed09f72c560e5302b3d98198b96bf1da635a4a","side":"right"},{"sibling":"d6607539da7ba39ec68be2d12f27ed6768766c745e3120fd915f88c5e288e07c","side":"right"},{"sibling":"b63408a424d27cd6a75e0fb155e69a58a328f41e9cb9dba1eddef9a5289cc7fd","side":"right"},{"sibling":"356fb36a4e188f03d7a05c54cd8789bdd40eda454b9bc9560f667acc08e6c4e0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":134886,"merkle_root":"c6c7ae28c065bced89e7f844216b073f1a7cc4b378db0d41a98bcd21b28066db","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260515T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-15T05:37:26Z","sig_algorithm":"ed25519","signature":"5a3978c26017daf4104adbb3e1c3099c5750157acbfc7242ece1815dc6740fe08a690291ce0afe42011e20cc565b5ebe64ec016bf658b5bab563a63337985c05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_16fe981b51950475c956e08f9938305d81b267b9a27a8753ded7bbcd8cacee41"}}