{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d6f891a07aa184220111fc1daf5d16496ef3da5b21aba44794653d82e5cae8c0","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d6f891a07aa184220111fc1daf5d16496ef3da5b21aba44794653d82e5cae8c0","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"fdae58175507ec12287b71446d624c2ed572ea4b9bcd4c6fbf00f277ff5504f2","published":"Tue, 21 Jul 2026 00:00:00 -0400","receipt_hash":"fdae58175507ec12287b71446d624c2ed572ea4b9bcd4c6fbf00f277ff5504f2","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"fdae58175507ec12287b71446d624c2ed572ea4b9bcd4c6fbf00f277ff5504f2","observed_at":"2026-07-21T04:43:35.036805Z","parent_run_hash":"03e944014de2697434479833d15ea9303e014945afc230ecc7f207824493b589","published":"Tue, 21 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.17560v1 Announce Type: new \nAbstract: Reinforcement learning (RL) provides a framework for sequential decision making under explicit objectives. In its classical form, RL studies how an agent should act to maximise long-term reward in a dynamic environment. In richer settings, the problem extends beyond a single agent and fixed environment: intelligent behavior may require strategic interaction, adaptation to uncertainty, and reasoning over high-dimensional worlds. This thesis studies RL from two perspectives: algorithms in games and RL in the era of foundation models.\n  The first part focuses on multi-agent RL in games. It examines how incentives, policies, and equilibrium concepts interact in competitive and general-sum environments, spanning two-player zero-sum games, large-scale video games, and multi-player settings with general structure. These works investigate learning in multi-agent systems and the behavior of RL methods in interactive environments. The second part ","title":"Reinforcement Learning: From Algorithms To Foundation Models","url":"https://arxiv.org/abs/2607.17560","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.17560v1 Announce Type: new \nAbstract: Reinforcement learning (RL) provides a framework for sequential decision making under explicit objectives. In its classical form, RL studies how an agent should act to maximise long-term reward in a dynamic environment. In richer settings, the problem extends beyond a single agent and fixed environment: intelligent behavior may require strategic interaction, adaptation to uncertainty, and reasoning over high-dimensional worlds. This thesis studies RL from two perspectives: algorithms in games and RL in the era of foundation models.\n  The first part focuses on multi-agent RL in games. It examines how incentives, policies, and equilibrium concepts interact in competitive and general-sum environments, spanning two-player zero-sum games, large-scale video games, and multi-player settings with general structure. These works investigate learning in multi-agent systems and the behavior of RL methods in interactive environments. The second part ","title":"Reinforcement Learning: From Algorithms To Foundation Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-21T04:43:35Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.17560"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1e0e54450221747f224e22f40aed1827c3b251f8f450043f46f86f30333868f296749c815d3cca41ca5a83023520553e6e1b6c5a9679917ccb37774bd1d2f401","signer":"crovia.substrate","subject":{"observed_at":"2026-07-21T04:43:35Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.17560"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c6fd124e04923bd292037f7f3506e78f916b5678e2fa83e327eb3a5618923174","leaf_index":336588,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"594a6d875c4b1c187be06a3fd377fa587efa2d234f699ece2646daa6a773c5e9","side":"right"},{"sibling":"d9ec850bec8c7412651b59d2737e9eb3a0f43e7f0674070add884ee6c4d5538c","side":"right"},{"sibling":"67e893dd942b6b8b696bc6249e12e98184c2efce95e149e22d0cfdd5e30367e7","side":"left"},{"sibling":"533f6864356a5fc5023677aae85e0c8850d4aa924e51e0265e9797c414232d04","side":"left"},{"sibling":"54c8abff7a167f90216e52f759babd710117bfd687ddcb2dbb7dc7cfdc78d433","side":"right"},{"sibling":"3af7f5627ece0b0aa3b8a46ebfe8c72367f02051656ed4864d6f665c93e28044","side":"right"},{"sibling":"78afb98e8815c9330b3a7e74c8b56112eb0284b980a66d5b96cce812169af054","side":"left"},{"sibling":"e6f9d6d6c760446a7b30dd4e30a28f58c0817f4bee09529ee009c470d86f5564","side":"left"},{"sibling":"38e5827f7c9f72ad34a2b97042f2fb5f7e868db899d074b7819af4f2209b9caa","side":"right"},{"sibling":"7b927551b5db06b6571913b4e6792ffcce5291a3eca5a0df4a3b6e296271105f","side":"left"},{"sibling":"b77a0b5ae4607c8fe6ba73449d46b35076e3dedc0c82a2c65a05780d42a7bc2e","side":"right"},{"sibling":"9eb5077edfb3dc553857d4794b925bfce117e0f8a1d049af5d0dd9026b470eef","side":"right"},{"sibling":"414b1a70fd1dcb25489a194714b97492b066684b15d0b7a48a176c4b9b5bc713","side":"right"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"613f015699131eb89bd755dee67133be95af25cf5f16c1c8ce4b99d963b8dd86","side":"right"},{"sibling":"a729b574b1135956436ded5eef1fe8f08014ff6a0729749d307ab1bca93fcdc9","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"4bf21052e085e8ac81f1dec1d2b310bd12bf948992de6177d12e9d2fda8d39f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":337144,"merkle_root":"5e969cc01afa67e4dbe5d37b712cdb10f4aa1fd74404e02eab724cf487c8d6d9","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260721T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-21T05:38:38Z","sig_algorithm":"ed25519","signature":"5c0c1a8dd2793d787ccd5e49e8b4d70eed136352555518589f05c74be357fc171702e42a76c3a556d90e3d51ff36cb3d292aaac83c66566de7b943f318bda50c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d6f891a07aa184220111fc1daf5d16496ef3da5b21aba44794653d82e5cae8c0"}}