{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_8c6f0ff69345fd7cae1324853670a29c966cc837bcfdf4d337a82f3afcbc2466","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_8c6f0ff69345fd7cae1324853670a29c966cc837bcfdf4d337a82f3afcbc2466","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"85845fbe32d3934a26f88ad52f7f1497efd5cc75edfff7a74f5e0b736f0d9c76","published":"Wed, 27 May 2026 00:00:00 -0400","receipt_hash":"85845fbe32d3934a26f88ad52f7f1497efd5cc75edfff7a74f5e0b736f0d9c76","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"85845fbe32d3934a26f88ad52f7f1497efd5cc75edfff7a74f5e0b736f0d9c76","observed_at":"2026-05-27T04:43:18.926230Z","parent_run_hash":"6f581915edab4326e2b95fed7c82c2ee149e978d6d7c2443439442a927c31dfa","published":"Wed, 27 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.26654v1 Announce Type: cross \nAbstract: Reinforcement learning (RL) often has a hierarchical structure, where an upper-level (UL) learner selects model parameters and a lower-level (LL) decision-making process responds, naturally leading to a bilevel optimization problem. Most existing bilevel RL methods assume a single-policy LL Markov decision process (MDP), and therefore fail to capture competitive structures arising in applications such as incentive design, where multiple policies interact. We study bilevel optimization problems in which the LL problem is a regularized min-max zero-sum Markov game and the UL objective is optimized through the saddle-point equilibrium induced by the LL game. In this work, we propose penalty-augmented Nikaido-Isoda descent-ascent (PANDA), a penalty-based first-order policy-gradient method based on the Nikaido-Isoda function. By exploiting the min-max game structure, PANDA avoids computing UL hypergradients and does not require second-order","title":"Bilevel Optimization over Saddle Points of Zero-Sum Markov Games","url":"https://arxiv.org/abs/2605.26654","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.26654v1 Announce Type: cross \nAbstract: Reinforcement learning (RL) often has a hierarchical structure, where an upper-level (UL) learner selects model parameters and a lower-level (LL) decision-making process responds, naturally leading to a bilevel optimization problem. Most existing bilevel RL methods assume a single-policy LL Markov decision process (MDP), and therefore fail to capture competitive structures arising in applications such as incentive design, where multiple policies interact. We study bilevel optimization problems in which the LL problem is a regularized min-max zero-sum Markov game and the UL objective is optimized through the saddle-point equilibrium induced by the LL game. In this work, we propose penalty-augmented Nikaido-Isoda descent-ascent (PANDA), a penalty-based first-order policy-gradient method based on the Nikaido-Isoda function. By exploiting the min-max game structure, PANDA avoids computing UL hypergradients and does not require second-order","title":"Bilevel Optimization over Saddle Points of Zero-Sum Markov Games","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-27T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.26654"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1f440dd7e81ab2e262cfff5e4a4828aa4555c42e1c1543f9d4e6c8197ccbb1fc45122432052dc4a3e2083b2adaeb341eef88bf2490a2e180a3f54fde10d0010e","signer":"crovia.substrate","subject":{"observed_at":"2026-05-27T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.26654"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d5a6a867c1a9a6082df469de882fbe2bdd06286885a9636e4245a3633becd887","leaf_index":153721,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1ce59967e37eed6b00ba7602c4ddf8a56cd0277414b5e7163d1c6a3b3df79036","side":"left"},{"sibling":"a0dc6a3de90f2d217b0c49d9b9ec0ec8255b80ecbcb7c4bd688a09264f7c3845","side":"right"},{"sibling":"4a9caf44fcd092819745daf4057d549cf68d6597fc5f5ae0f61b01bf65a4d373","side":"right"},{"sibling":"d27190d404dbf25c0c5766697001fbe48a9d0593f97709319385c64b15ac1b20","side":"left"},{"sibling":"cb5a0f85e5be6f161846ef96c8b99b4bd57c7308b2dbd6384516699e6682bbe2","side":"left"},{"sibling":"1083d57987922442c2f8f10fa547e0641aebbd2015a20a97f041d0a6f4375e48","side":"left"},{"sibling":"727ac1dfc33933f05b7eed6f887414eb57d6283ba0cbf0d562eadb6f86c3d644","side":"left"},{"sibling":"e6fcc8fd02ff764f6e74465d070f84e313e858647d901e3ee00b83931106f26c","side":"right"},{"sibling":"c56e46802f42ead6bcf79619c80428b6dc22e9afc9611a87f0d04a5ef9acadd7","side":"right"},{"sibling":"3165125427a29042fc9d02858a59a68858dbdcb2e19d1afa5f6dd6a95cfbce6a","side":"right"},{"sibling":"04b9a68b8ec6fa37251564383c685c23ce69e5e031df4eae69f79a3a334b68bf","side":"right"},{"sibling":"816f233274bb10f5a122aac086a0c8c697b78fec67a4af55190bb596b7506fab","side":"left"},{"sibling":"d415e6939aee710631f5062799379b547d2c3e3d9a68f263bbb5a693285ab2ca","side":"left"},{"sibling":"374c02d15fb12bd356c179c94766043a982052c6132af8bfc15361b431ffa9f7","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"990705096edf483cc877217308f731dc42d6f6d99f82880167bbdbbfef32560a","side":"right"},{"sibling":"dd265753d95fa2e2fb4f5768e37fab6f691ccff09ad60d0910ff7dc23bac9226","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":154065,"merkle_root":"4993cfdc172e7880b60667f16789dc2e831ff000f81bb1ecba248e73f1510eca","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260527T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-27T05:37:36Z","sig_algorithm":"ed25519","signature":"76ecf118011540405e96506e6219752df04a2850632f2903dc6f10e08b98bc5a42c8d9e1cb5b7c5bf714479806a403df5f34399afa40c23fbb71493a1f77bd0c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_8c6f0ff69345fd7cae1324853670a29c966cc837bcfdf4d337a82f3afcbc2466"}}