{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_15f22a0943e84466d7e049d07d4d7b2e62533498e731f3526d06550dbd78d5ef","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_15f22a0943e84466d7e049d07d4d7b2e62533498e731f3526d06550dbd78d5ef","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8beb98397ec95b0d8372f227d7a4ce42cb060e7faa983cb5b966d22dca727bf3","published":"Wed, 17 Jun 2026 00:00:00 -0400","receipt_hash":"8beb98397ec95b0d8372f227d7a4ce42cb060e7faa983cb5b966d22dca727bf3","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8beb98397ec95b0d8372f227d7a4ce42cb060e7faa983cb5b966d22dca727bf3","observed_at":"2026-06-17T04:43:17.968423Z","parent_run_hash":"8f56c4deb22b28178ba7974d6ffc5ff17336d44c95dd94de80095705245fa113","published":"Wed, 17 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.17735v1 Announce Type: new \nAbstract: Although reinforcement learning (RL) has expanded the cognitive boundaries of large language models (LLMs), it often remains vulnerable to the autoregressive curse in long-horizon logical reasoning: small epistemic perturbations introduced early in generation can propagate irreversibly along the Markov decision process flow, triggering cascading failures that drive the reasoning trajectory toward collapse. To overcome this autoregressive cascade, in which a single early mistake can compromise all subsequent reasoning steps, we propose dynamic epistemic entropy orchestrated erasable reinforcement learning ($\\text{E}^3\\text{RL}$). $\\text{E}^3\\text{RL}$ eliminates reliance on external signals by grounding the model's endogenous local autoregressive cross-entropy as an intrinsic coordinate of epistemic uncertainty. By introducing segment-level adaptive dynamic thresholds and advantage allocation, $\\text{E}^3\\text{RL}$ enables the model to pr","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","url":"https://arxiv.org/abs/2606.17735","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.17735v1 Announce Type: new \nAbstract: Although reinforcement learning (RL) has expanded the cognitive boundaries of large language models (LLMs), it often remains vulnerable to the autoregressive curse in long-horizon logical reasoning: small epistemic perturbations introduced early in generation can propagate irreversibly along the Markov decision process flow, triggering cascading failures that drive the reasoning trajectory toward collapse. To overcome this autoregressive cascade, in which a single early mistake can compromise all subsequent reasoning steps, we propose dynamic epistemic entropy orchestrated erasable reinforcement learning ($\\text{E}^3\\text{RL}$). $\\text{E}^3\\text{RL}$ eliminates reliance on external signals by grounding the model's endogenous local autoregressive cross-entropy as an intrinsic coordinate of epistemic uncertainty. By introducing segment-level adaptive dynamic thresholds and advantage allocation, $\\text{E}^3\\text{RL}$ enables the model to pr","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-17T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.17735"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:0fd99227a65118f5dc8c4c6030c3cc0e27f7f5772abda8ced2d963e402045fef30a19eed4b7b2e8b089cdbf96d4b66f5ddf5114f0567940be285d3096f835003","signer":"crovia.substrate","subject":{"observed_at":"2026-06-17T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.17735"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"04e87923421bf2b99f4497437cb2b039e4098f55e34c6d01838a4c6fce7e7ad8","leaf_index":230874,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5d43daa1fc2876f112c6c7b9fef668419c979d4967e7ecfbe8c4b785e85e9a92","side":"right"},{"sibling":"571867d40421bbd201f046ece85bcb1e60fab5761fe548875ebedf3a8ac90d7e","side":"left"},{"sibling":"284a0e178fb4e7b1b6fba5d36f7d988451a7e4ae2c6659ab69aa458c864e6dd8","side":"right"},{"sibling":"38054516521a1afb51ef23edcffe1a9375c95d0095401f8961a9a0ebe166e09e","side":"left"},{"sibling":"a011eed65a011c5f09f35b365315daa3efe52202fe1f717062f18edb32070310","side":"left"},{"sibling":"505980e02b3a22a962e22e6151b370e9aa9ce58f596e99b446b88c8236da18aa","side":"right"},{"sibling":"70b54557b0f44aae58967d06c700194d1d2cf17d2bbbb2ad9632da8cf5919c79","side":"left"},{"sibling":"46f3925bb1c995570704f74c45315769d7303b87cc2d1f176bed247b9cc9fdad","side":"left"},{"sibling":"ee59602dad0bf74c97a32a93f0a1a19e7a12f2800791988a6fdf35611febe031","side":"left"},{"sibling":"0c5669692381d605223c74b8d30f70cd308e77e33d5e40ea84bb7b4f84f2d4d9","side":"right"},{"sibling":"d5b9f8b1a2c9f6a46e17982dfbe6ce1f3b5fa4e730220397f2253d114dcc8486","side":"left"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_15f22a0943e84466d7e049d07d4d7b2e62533498e731f3526d06550dbd78d5ef"}}