{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_ffd3f082858e3a9752feb3f8c2be51e5e0f70c9d5658f95621f29a6cb5abf711","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_ffd3f082858e3a9752feb3f8c2be51e5e0f70c9d5658f95621f29a6cb5abf711","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"5f390f97a9a6e9e0a94890b1ae51d6b20bcb27e09fe153dc994c3b61e938f70b","published":"Fri, 29 May 2026 00:00:00 -0400","receipt_hash":"5f390f97a9a6e9e0a94890b1ae51d6b20bcb27e09fe153dc994c3b61e938f70b","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"5f390f97a9a6e9e0a94890b1ae51d6b20bcb27e09fe153dc994c3b61e938f70b","observed_at":"2026-05-29T04:43:58.478092Z","parent_run_hash":"0fcd87efcfe67ccb9952f747541debc16793919a4d20fd71ca0ad5516a0a13ee","published":"Fri, 29 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2509.23730v2 Announce Type: replace \nAbstract: Large language models (LLMs) have recently advanced in reasoning when optimized with reinforcement learning (RL) under verifiable rewards. Existing methods primarily rely on outcome-based supervision to strengthen internal LLM reasoning, often leading to inefficient exploration and sparse rewards. To mitigate this issue, we propose Expert-Assisted Policy Optimization (EAPO), a novel RL framework that enhances exploration by incorporating multi-turn interactions with external experts during training. Unlike prior methods, where policies reason in isolation, EAPO incentivizes the policy to adaptively determine when and how to consult experts, yielding richer reward signals and more reliable reasoning trajectories. External assistance ultimately internalizes expert knowledge into the policy model, amplifying the model's inherent reasoning capabilities. During evaluation, the policy model has been well-optimized to solve questions indepe","title":"EAPO: Enhancing Policy Optimization with On-Demand Expert Assistance","url":"https://arxiv.org/abs/2509.23730","vendor":"arxiv_cs_ai"},"summary":"arXiv:2509.23730v2 Announce Type: replace \nAbstract: Large language models (LLMs) have recently advanced in reasoning when optimized with reinforcement learning (RL) under verifiable rewards. Existing methods primarily rely on outcome-based supervision to strengthen internal LLM reasoning, often leading to inefficient exploration and sparse rewards. To mitigate this issue, we propose Expert-Assisted Policy Optimization (EAPO), a novel RL framework that enhances exploration by incorporating multi-turn interactions with external experts during training. Unlike prior methods, where policies reason in isolation, EAPO incentivizes the policy to adaptively determine when and how to consult experts, yielding richer reward signals and more reliable reasoning trajectories. External assistance ultimately internalizes expert knowledge into the policy model, amplifying the model's inherent reasoning capabilities. During evaluation, the policy model has been well-optimized to solve questions indepe","title":"EAPO: Enhancing Policy Optimization with On-Demand Expert Assistance","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-29T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2509.23730"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:be2a2b2db8ca6ec0fa63153a05e2333434a035b24d01f74e2bc40b4da0c9323b3a67f004682a0e91cb1d83d8f4ac2ec92c5f3022f26ccf1f78c642931fed8d0f","signer":"crovia.substrate","subject":{"observed_at":"2026-05-29T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2509.23730"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c431d03cef0e3af4165745f34afe5801446958eaa177059e08590ea297a20863","leaf_index":158002,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1326cba323802c7ba442eea352cbe31262f5ea231c0f5dd4b1ef520b9db56cef","side":"right"},{"sibling":"edd93dcfe60d6c057dc5741bca977894754ca592dc6a12deeed184f53385b0f4","side":"left"},{"sibling":"637d6112ccac7187e41d573c92d5be965110e967c7989788be6408173021389b","side":"right"},{"sibling":"68a671e5d6047e684d4f7753bacf32e3cc94f0e1d5ce07c235b38f51ade27bf1","side":"right"},{"sibling":"a821abbb8f8cc3dd8a07c1262d29e28c496b4eea05be2e2d201121ed7d075bed","side":"left"},{"sibling":"fea7174f0a136a206dde2903ff85ee04d9ef78a16cc9ccf7baf52adfad38ec9a","side":"left"},{"sibling":"6ec9548fe4b05de3b418873e8f539457b0d358597426bbc33615ea29f980e57e","side":"right"},{"sibling":"7d396b40cab1f5921441ef7dd8b2f11230a90a7ab021a73b9ae8fa5ac45029fa","side":"right"},{"sibling":"002471ff5893cd2a2890119cbfe4a9456bfb46cf31fa6aee377650afe6ae0c92","side":"left"},{"sibling":"1dcc44e23fbb0218b13591e4b584eca3600dcf365769cb741e0ecd33b25b8c56","side":"right"},{"sibling":"fcf16a6f44025801f5b83e928acce764352ddbc06ae3043d8cde9a409933e6d8","side":"right"},{"sibling":"78982294dee68f9db9288c64d7e507c7865fda96e1b6f7ccff5c8bb152e93c49","side":"left"},{"sibling":"995b421824624a8282c7f44e64c64ee35344800f477ae1845b41be14d3fab94c","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"eef0e8906a749d3470f89beeedc723f37a5737010bbb0dcc7cf91515338e5a3e","side":"right"},{"sibling":"1a07e481a9407d71aad078ce854cdeee362163c887fe10f889b0ecf0b5e749ad","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":158251,"merkle_root":"485e6b31fe60c8beba5b394808c7e4c32448b2ff65c2482c480ca0e2a2eda718","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260529T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-29T05:37:37Z","sig_algorithm":"ed25519","signature":"bbf9f005201182fce4f9d94c7a9d01a508b56daf7d9611bd73514f5f616bc059d0e5e1f2edfc95716e6fe08ef5fae38b558cbf7f2fd8f9d5dfe4c34a54c83005","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_ffd3f082858e3a9752feb3f8c2be51e5e0f70c9d5658f95621f29a6cb5abf711"}}