{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_58574fb428754b0df5436814f8ed2ce604c19f433364ce7f195dfb57916f4fb0","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_58574fb428754b0df5436814f8ed2ce604c19f433364ce7f195dfb57916f4fb0","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e656ad15e07a701d8620d1d7dd524ae66594546e2c7d8bcd5b8a473a1e08f683","published":"Wed, 03 Jun 2026 00:00:00 -0400","receipt_hash":"e656ad15e07a701d8620d1d7dd524ae66594546e2c7d8bcd5b8a473a1e08f683","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e656ad15e07a701d8620d1d7dd524ae66594546e2c7d8bcd5b8a473a1e08f683","observed_at":"2026-06-03T04:43:57.136784Z","parent_run_hash":"62ae9c8eda846b00bc49666345b338d00756c5203438666c4c1fc694cc364b84","published":"Wed, 03 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.03762v1 Announce Type: cross \nAbstract: Agentic reinforcement learning (RL) equips large language models (LLMs) with tool-use capabilities that substantially improve reasoning on complex tasks. However, integrating external tools often destabilizes training: over-reliance on tools can induce input distribution shift, while overly conservative tool use limits effective exploration. To address this issue, we propose a unified framework TAO-RL that couples tool-aware trajectory filtering with entropy-guided exploration for efficient policy optimization. Specifically, at the data level, TAO-RL filters rollout trajectories along two criteria: discarding those where all tool invocations fail to execute, and removing those where all rollouts are either correct or incorrect, as both cases yield degenerate advantage estimates that contribute no discriminative learning signal. This joint filtering retains data that are both tool-capable and informative, establishing a high-quality tra","title":"Tool-Aware Optimization with Entropy Guidance for Efficient Agentic Reinforcement Learning","url":"https://arxiv.org/abs/2606.03762","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.03762v1 Announce Type: cross \nAbstract: Agentic reinforcement learning (RL) equips large language models (LLMs) with tool-use capabilities that substantially improve reasoning on complex tasks. However, integrating external tools often destabilizes training: over-reliance on tools can induce input distribution shift, while overly conservative tool use limits effective exploration. To address this issue, we propose a unified framework TAO-RL that couples tool-aware trajectory filtering with entropy-guided exploration for efficient policy optimization. Specifically, at the data level, TAO-RL filters rollout trajectories along two criteria: discarding those where all tool invocations fail to execute, and removing those where all rollouts are either correct or incorrect, as both cases yield degenerate advantage estimates that contribute no discriminative learning signal. This joint filtering retains data that are both tool-capable and informative, establishing a high-quality tra","title":"Tool-Aware Optimization with Entropy Guidance for Efficient Agentic Reinforcement Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-03T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.03762"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b4d544b6775d5008510ea989c7040c7db3fc8c858f7311bf3f0e9225a243b2b37f23677cf4605b1bade3ba0d73ac5fc88f2b11fe42a118f197930131cc117304","signer":"crovia.substrate","subject":{"observed_at":"2026-06-03T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.03762"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"769130ccc56502473023b86b2a6e05202eb247c365c4176308db972906d2e993","leaf_index":209259,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"87c45dde824e8200036dd05d0e6753fd0b1ac8bc513422c56740d9a87dfc8159","side":"left"},{"sibling":"8d6ac93163792a8e68370f3644931a92d92b00d101324ad54d72c64f1874d9e3","side":"left"},{"sibling":"b694b1f899436a35a5fa3732b8827f55c95fb4f3e82837f4c66ef55127e7aeea","side":"right"},{"sibling":"c0c5ea1ab5756fd4cc5d3424a333fd12291eea906f3026cc5ee81e1f27416e65","side":"left"},{"sibling":"33ecb553c3eb3aa7c6bbc5a248aa84370002dc3cbbf716cf4b4afe9e96cf7e4e","side":"right"},{"sibling":"16d6c8efeec2fb8581ca5df04bbba0eaf0b84a88b42c35aaa3f5e1d5b6698b51","side":"left"},{"sibling":"801b1ba88a919eeea0ffe1d64b4aaf9af369587100583a6d3683ef9572bb5fe1","side":"left"},{"sibling":"2ef1f1a576d002b7a433f7807e3e3de79248e5d1145c4394fef6eb45b8063603","side":"right"},{"sibling":"2b9867040ec52d22722edc26b204e15651723dcc66d361cc1af11490a854d461","side":"left"},{"sibling":"2abcec6d14b256f82b270a3141886de6129141dec525543607b760f02588a877","side":"right"},{"sibling":"2b6b45743f97ac502854e489ac38a3366f8ae7ede58a2728b45daadbf29e9d03","side":"right"},{"sibling":"a81babbd79ea0da9e030dd7f43bffb6519d214317decd50727bea4e78189d970","side":"right"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"8d3baa674a45fd8bc6d3e8d25298f4bec86c72fa8259d576c330d12955c7b4f7","side":"right"},{"sibling":"e32819d1eff909db08066d1703f2db3f091cddac19378b2c0c625ab11e3fdbc0","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":209569,"merkle_root":"c852efc8ad7dfffc196c71380f79e6398bcaf566974cbae7c9950b0f600d5bc8","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260603T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-03T05:37:57Z","sig_algorithm":"ed25519","signature":"1ca417607effc813341721cfdade0a2d2d4ba96a90d361dbc3e864b6991b90abc326ec41d9ba3e7110cc04adb29d7b8d95abea7ff2ab8c591b2f864ed7270c03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_58574fb428754b0df5436814f8ed2ce604c19f433364ce7f195dfb57916f4fb0"}}