{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_4489e0e77779cad29dafb281c846d9078a6ade4854875c71e2be27a62466a48c","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_4489e0e77779cad29dafb281c846d9078a6ade4854875c71e2be27a62466a48c","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"5ec7044a51e3f7b056c45b571e0d52d75a5e46412a0695548c09b6161ad1704b","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"5ec7044a51e3f7b056c45b571e0d52d75a5e46412a0695548c09b6161ad1704b","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"5ec7044a51e3f7b056c45b571e0d52d75a5e46412a0695548c09b6161ad1704b","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.00017v1 Announce Type: new \nAbstract: Training language model agents for multi-agent strategic interaction presents a core difficulty: the quality of any action may depend on future events that never materialize, on moves that violate game rules, or on decisions made by other players. Standard reinforcement learning assumes that rewards can be assigned at each step, but this assumption fails in settings where outcomes are entangled across time and agents. We introduce delayed per-step reward attribution with eligibility gating, an episode lifecycle and postprocessing pipeline that computes rewards only at episode end, propagates them back to originating steps according to task-specific semantics, and excludes steps that lack valid dependent information from training. Together with asynchronous rollout generation via vLLM's continuous batching, curriculum-based opponent sampling, and multi-level stratified batch construction, this approach enables stable, sample-efficient RL ","title":"MindGames Arena Generalization Track: In2AI Solution with Delayed Per-Step Reward Attribution","url":"https://arxiv.org/abs/2606.00017","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.00017v1 Announce Type: new \nAbstract: Training language model agents for multi-agent strategic interaction presents a core difficulty: the quality of any action may depend on future events that never materialize, on moves that violate game rules, or on decisions made by other players. Standard reinforcement learning assumes that rewards can be assigned at each step, but this assumption fails in settings where outcomes are entangled across time and agents. We introduce delayed per-step reward attribution with eligibility gating, an episode lifecycle and postprocessing pipeline that computes rewards only at episode end, propagates them back to originating steps according to task-specific semantics, and excludes steps that lack valid dependent information from training. Together with asynchronous rollout generation via vLLM's continuous batching, curriculum-based opponent sampling, and multi-level stratified batch construction, this approach enables stable, sample-efficient RL ","title":"MindGames Arena Generalization Track: In2AI Solution with Delayed Per-Step Reward Attribution","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.00017"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a23613cb89f09dbc13012109f9db887e3a2d5d3448e9983d175aefc12210ce4d594711c6e63452fd3d9117e945b1aae8db8ec45c215d29091f615cc745f50001","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.00017"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"965c323e1d510447ebf6ac796f2dbdc9f8cc88d3d7067b5e8353caeead216447","leaf_index":205166,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"ea4bfbcc8d9f7f739cfcde6e970cf919c1901679d263ddb1f04e63da69fad3e7","side":"right"},{"sibling":"3413be4a7bc78574ac5e9839ce7d1c2003084ee0c7e1e47b50035fdddd8be0aa","side":"left"},{"sibling":"c3b73a7ed8acdd1ad893ca0b0ea624325f9df3a69e51ddef19eb51b6ac54b7e1","side":"left"},{"sibling":"c64d639a46d72d85abaec4be4f08e44b0056831e50bd4326f5a03189ee61b1cb","side":"left"},{"sibling":"4e41fa2c94f3237f98a5ed104133541d4965faaa8ecd57dac02b1cfb7f51e3e9","side":"right"},{"sibling":"71507214c98c5c6d2cbd0eda08ce54ea4a84e232295243c4f52c00df3fa8ee90","side":"left"},{"sibling":"ff234fd55a4c32ca6bd14fa67c40b4f4eb2aeb8f7dc7f4172cb03f8f9e95b244","side":"left"},{"sibling":"960bf3f7015f8979c4dde25edf3b1e2e5a86134d3d950b4d26f5761ce27b0275","side":"right"},{"sibling":"3fed4154aebacb68ca58546a945dace8a8d7d2176af23d8fb700f153f2df448f","side":"left"},{"sibling":"30520dbc0b3dbdd0b4502a5ecc782d8aa02c8659f12dfb9b79fd55de5c05a0c5","side":"right"},{"sibling":"a82575bfb494af7afcc13aaae718afa6f74030f09d71b02819ea25efd4186fc4","side":"right"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_4489e0e77779cad29dafb281c846d9078a6ade4854875c71e2be27a62466a48c"}}