{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_5ebe461399afeb7abefe275abf9bea3ef361de525517ac90bbcd7847162413fd","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_5ebe461399afeb7abefe275abf9bea3ef361de525517ac90bbcd7847162413fd","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8c475976708237856a0f895aba1cce854759435872fb5f8b2face96e71ac8774","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"8c475976708237856a0f895aba1cce854759435872fb5f8b2face96e71ac8774","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8c475976708237856a0f895aba1cce854759435872fb5f8b2face96e71ac8774","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2603.12109v2 Announce Type: replace \nAbstract: Reinforcement learning (RL) has become a de facto paradigm for building LLM-based agents that act, interact, and reason over extended task horizons. However, in active reasoning where agents must elicit new observations through interaction with the environment to solve the task, we find that outcome-based RL can induce a systematic failure mode which we call information self-locking (SeL): agents fail both to elicit informative feedback and to internalize obtained evidence. To understand the issue, we trace agentic behaviors into two coupled capabilities: Action Selection (AS), which determines observation streams, and Belief Tracking (BT), which updates the agent's internal task understanding. Theoretical and empirical analyses reveal a bidirectional bottleneck that leads to SeL: weak BT obscures the credit of informative actions, while weak AS deprives BT of useful evidence. This coupling weakens the learning signal for both capabi","title":"On Information Self-Locking in Reinforcement Learning for Active Reasoning of LLM agents","url":"https://arxiv.org/abs/2603.12109","vendor":"arxiv_cs_ai"},"summary":"arXiv:2603.12109v2 Announce Type: replace \nAbstract: Reinforcement learning (RL) has become a de facto paradigm for building LLM-based agents that act, interact, and reason over extended task horizons. However, in active reasoning where agents must elicit new observations through interaction with the environment to solve the task, we find that outcome-based RL can induce a systematic failure mode which we call information self-locking (SeL): agents fail both to elicit informative feedback and to internalize obtained evidence. To understand the issue, we trace agentic behaviors into two coupled capabilities: Action Selection (AS), which determines observation streams, and Belief Tracking (BT), which updates the agent's internal task understanding. Theoretical and empirical analyses reveal a bidirectional bottleneck that leads to SeL: weak BT obscures the credit of informative actions, while weak AS deprives BT of useful evidence. This coupling weakens the learning signal for both capabi","title":"On Information Self-Locking in Reinforcement Learning for Active Reasoning of LLM agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2603.12109"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:03cd50301ccc6b0872ca30c88ab2cd2f458653ff8c90ff0413105506dc010de08d3c7f75f75b42c546552c36725bf618f4dae32617dc5aeee241d4f399a48d02","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2603.12109"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"9de2a710042ebb430d2fdebaa352b7f03c45b70468949f48d5c26c439653ccfe","leaf_index":205781,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"e3d627ee298972b49d880a6f5acb66536271539985deb77ec28fa7e233cdb685","side":"left"},{"sibling":"a70361074aa38723893ef134d3722e38c21c6475c444a7995c6d2cd544ca262d","side":"right"},{"sibling":"53d1c9cf6601b243b4a6f47072a45da9aacd5c781358c353b7d6b51b65ee6e1a","side":"left"},{"sibling":"d6ffa4804b38df037c5318afd088a4da0285b95873e9e220298af2214feb2c32","side":"right"},{"sibling":"4a2b5dcba81d8f7b1bf826bf32b5fb5187ba3293dcba90facfb4ed4d056ba027","side":"left"},{"sibling":"bcfd856087d97e6283a4cbf68b3105c0fa1c9171338ef74f6e6fb82633a1d8db","side":"right"},{"sibling":"199f39173936d8df4dd145829492a9dceaf32da27938b47dffcfee0b9a047c55","side":"left"},{"sibling":"f90a21898085db59f0f27c56200706376a54d0bca343ed9e47ede5df384cd547","side":"left"},{"sibling":"21332ee1c3955bf1803955bf945ab0966ce5e0aa48d8642320cdbc044a72b935","side":"left"},{"sibling":"6620a5008acf732cd3e57b0f2d1437293e88366c2e5b175ebeb7d796fc1e0c62","side":"left"},{"sibling":"a82575bfb494af7afcc13aaae718afa6f74030f09d71b02819ea25efd4186fc4","side":"right"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_5ebe461399afeb7abefe275abf9bea3ef361de525517ac90bbcd7847162413fd"}}