{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_61e6f35909efcc0b97402742df078e3a69ec4a283e19838418c6f5aff4a4170d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_61e6f35909efcc0b97402742df078e3a69ec4a283e19838418c6f5aff4a4170d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"2ac4568c4439404db5c2b29b49d883de54731ad5935b40c6182714c8a0f35093","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"2ac4568c4439404db5c2b29b49d883de54731ad5935b40c6182714c8a0f35093","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"2ac4568c4439404db5c2b29b49d883de54731ad5935b40c6182714c8a0f35093","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.04713v1 Announce Type: cross \nAbstract: Reinforcement learning holds significant potential for training large language models (LLMs) to handle multi-turn interactive tasks. However, in long-horizon, multi-turn tasks characterized by sparse outcome rewards, directly training with outcome rewards often results in slow convergence due to the sparsity of signals and the lack of fine-grained feedback. Furthermore, the model may fail to learn successful trajectories that are not sampled during training, thereby limiting its performance. Conversely, while employing customized dense process rewards provides richer signals and accelerates convergence, these surrogate rewards may exhibit potential misalignment with the ground-truth outcome rewards. This inconsistency can bias the training direction and ultimately degrade the model's final performance. In this work, we propose Reward-Swap Policy Optimization (RSPO), a method designed to leverage the rich information from dense process ","title":"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents","url":"https://arxiv.org/abs/2607.04713","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.04713v1 Announce Type: cross \nAbstract: Reinforcement learning holds significant potential for training large language models (LLMs) to handle multi-turn interactive tasks. However, in long-horizon, multi-turn tasks characterized by sparse outcome rewards, directly training with outcome rewards often results in slow convergence due to the sparsity of signals and the lack of fine-grained feedback. Furthermore, the model may fail to learn successful trajectories that are not sampled during training, thereby limiting its performance. Conversely, while employing customized dense process rewards provides richer signals and accelerates convergence, these surrogate rewards may exhibit potential misalignment with the ground-truth outcome rewards. This inconsistency can bias the training direction and ultimately degrade the model's final performance. In this work, we propose Reward-Swap Policy Optimization (RSPO), a method designed to leverage the rich information from dense process ","title":"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.04713"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b3f2b7a53ec91d2898b98a59129be37c25c09f90b906b51d0199197fa966b61a8ac6623397fa959bd4545070828a20b5442eef5f308c50d4127631423d808f01","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.04713"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7b3fd7df2d3f7d1905d6b11758f784a07575ef92d92c60d5235a7868d6ace12b","leaf_index":289097,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"de8629d6299f4815c538f6cf71a8aef0be9bd134fe033dbed986a4d3f908f680","side":"left"},{"sibling":"ae7b6955f4e3c28b675b90e1f033c75e450cd873ac80538f2012cc4f99fba421","side":"right"},{"sibling":"7fa1be9b2efcc8bc8a09f4718b975ea51fc5fe4939b7f43e5ee32cf7c2df0c6b","side":"right"},{"sibling":"6e4e28f38bf015b75aa4807142da99c95fece57c7d38a5ed526fa3b48c47f8fd","side":"left"},{"sibling":"689aa3eed06c2416306377d26d38614a3d616d28953c438bd5f6a6516df018ec","side":"right"},{"sibling":"dedfa0341e330850adec7653f1402ddaf62e87c235be6120547dd8365bf29160","side":"right"},{"sibling":"7aeacee516154284940b318d4c22233fbe3a13ffa98f9f849efdab541f44dc2b","side":"left"},{"sibling":"3b53d97c225577b0a2d53eeb1ca093f0c3441edf0f64e622c325f18db4fae31c","side":"right"},{"sibling":"35d3e8088ee93171aa479055dcd001cfb6d23925114ddc0908531e54298b0d29","side":"left"},{"sibling":"841129c21a7583176cdc7de281cadfe0e00d04673461e199760cd5128d8cc2d5","side":"right"},{"sibling":"19d6dfd29bc47f35fa02e8fe765277ba9cc3e6da5072309f24ebaac5b5f295e3","side":"right"},{"sibling":"8e0ad7889eb2d4b40e5b6c3d8e2eb19d4e202374983f468aa76321823de07a9f","side":"left"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_61e6f35909efcc0b97402742df078e3a69ec4a283e19838418c6f5aff4a4170d"}}