{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c4cd987f171c1729ad02308f9c166530176829326663039b5eb9501c183ce61b","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c4cd987f171c1729ad02308f9c166530176829326663039b5eb9501c183ce61b","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"d624e1572deb0b25e67969e53f626e3b43eec654b52b64c48db8485c6d5c5922","published":"Thu, 04 Jun 2026 00:00:00 -0400","receipt_hash":"d624e1572deb0b25e67969e53f626e3b43eec654b52b64c48db8485c6d5c5922","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"d624e1572deb0b25e67969e53f626e3b43eec654b52b64c48db8485c6d5c5922","observed_at":"2026-06-04T04:43:08.243501Z","parent_run_hash":"298818240313a3c9ce3ace3750dbf55845013dc3bc59aeced6331eccefdb61ac","published":"Thu, 04 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.11510v2 Announce Type: replace-cross \nAbstract: To encourage diverse exploration in reinforcement learning (RL) for large language models (LLMs) without compromising accuracy, we propose Policy Split, a novel paradigm that bifurcates the policy into normal and high-entropy modes with a high-entropy prompt. While sharing model parameters, the two modes undergo collaborative dual-mode entropy regularization tailored to distinct objectives. Specifically, the normal mode optimizes for task correctness, while the high-entropy mode incorporates a preference for exploration, and the two modes learn collaboratively. Extensive experiments demonstrate that our approach consistently outperforms established entropy-guided RL baselines across various model sizes in general and creative tasks. Further analysis reveals that Policy Split facilitates dual-mode exploration, where the high-entropy mode generates distinct behavioral patterns to the normal mode, providing unique learning signals","title":"Policy Split: Incentivizing Dual-Mode Exploration in LLM Reinforcement with Dual-Mode Entropy Regularization","url":"https://arxiv.org/abs/2604.11510","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.11510v2 Announce Type: replace-cross \nAbstract: To encourage diverse exploration in reinforcement learning (RL) for large language models (LLMs) without compromising accuracy, we propose Policy Split, a novel paradigm that bifurcates the policy into normal and high-entropy modes with a high-entropy prompt. While sharing model parameters, the two modes undergo collaborative dual-mode entropy regularization tailored to distinct objectives. Specifically, the normal mode optimizes for task correctness, while the high-entropy mode incorporates a preference for exploration, and the two modes learn collaboratively. Extensive experiments demonstrate that our approach consistently outperforms established entropy-guided RL baselines across various model sizes in general and creative tasks. Further analysis reveals that Policy Split facilitates dual-mode exploration, where the high-entropy mode generates distinct behavioral patterns to the normal mode, providing unique learning signals","title":"Policy Split: Incentivizing Dual-Mode Exploration in LLM Reinforcement with Dual-Mode Entropy Regularization","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-04T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.11510"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:aad975b132b629bdf83098a59a996ff38934ab34d696bf73ab90d112ab4de2aeeb377e07f745fcd6a22b13f541b262f2edcee80e747651dce6db2f6c21d1b206","signer":"crovia.substrate","subject":{"observed_at":"2026-06-04T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.11510"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d8c63a5fc00562d192dbc6fadbaf1325af5c9d4d1c3b424c9ccc73f2000562b7","leaf_index":212875,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fe5e7b3ff796eb981907a7b69ce22a0235a11c4658a0621dd2375fe9d4e13d82","side":"left"},{"sibling":"7ec4571dfca37e679d93c1be0f7546830ebfc315143375ef026ee1d5b621a598","side":"left"},{"sibling":"2a8398553f7fe55083d3bc93f194fa822e8de6671b63df68cb01dfaa86fa727f","side":"right"},{"sibling":"02738a163fe36fd1f24f4092cfd8607697a9cd1a88f28b0b43eb0e8cedc9dfaf","side":"left"},{"sibling":"d3d71226581290ed6878782c16d159efec8cfbbb44f484e60c94614763205dd1","side":"right"},{"sibling":"279eaf93fa03da75e98e85e0498ea45dc911f2fa8496f288bd8e8393b76e43a5","side":"right"},{"sibling":"d3e42a6c46328ee67e86cd8ee47af0826f2c502d9570e347a543ab25d8fd2fee","side":"right"},{"sibling":"8e52d8857251092c8d86330e4f43c6472257636abff3006310489f2170e2b3e3","side":"left"},{"sibling":"b4e8a5af3a9fdc4bb815270cd541ee4baf09fb1d9e2cccd75c7346558e1f1e4c","side":"left"},{"sibling":"cfb460164a914d1f96a36aa17124b45bb5417829d2adbe5ca48375dbd842ec44","side":"left"},{"sibling":"a84ebc8e894a9893ee34d1afc942d15f28243b926f1e5f9d7f10a3de1e795262","side":"left"},{"sibling":"f8f6bd9da448fa097e2115b71146f61691d9f8807aca291c87c633192a7224e9","side":"left"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"422bcf7e281ca3a607f3726a5e6b8fabdb85e8b32199b4a86356998e260a0b34","side":"right"},{"sibling":"54a99163a4a62374c3ca6fb46294222f1d4b1a9d0b636e27256b0e093e98239a","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":213053,"merkle_root":"19d104b92c4d7299881c447fb8422611fc9cb8615d37a538959e34a2da7ef55f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260604T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-04T05:37:49Z","sig_algorithm":"ed25519","signature":"630748e88645187aa3b4d4cb8c872cc146180d0301e4c6656f15f99081d7776432715cea2a997b32b7b3a5216f215d19c1b536c0b093100ec843fa0bd65f2101","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c4cd987f171c1729ad02308f9c166530176829326663039b5eb9501c183ce61b"}}