{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1d268ca1f183a0a12148483a4d5dc954f69fe913cac241c41e3401795a290b24","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1d268ca1f183a0a12148483a4d5dc954f69fe913cac241c41e3401795a290b24","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0c7e27b441cf33a36c53c08b0733eacd0e5223c62c35b7ae9172f40daa331f09","published":"Thu, 18 Jun 2026 00:00:00 -0400","receipt_hash":"0c7e27b441cf33a36c53c08b0733eacd0e5223c62c35b7ae9172f40daa331f09","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0c7e27b441cf33a36c53c08b0733eacd0e5223c62c35b7ae9172f40daa331f09","observed_at":"2026-06-18T04:43:37.219665Z","parent_run_hash":"de79a40f7b3537d88842f7ac355e799c5df2adcb4fc32a4e28096d4bbdf01739","published":"Thu, 18 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.19236v1 Announce Type: cross \nAbstract: Reinforcement Learning with Verifiable Rewards algorithms like GRPO have emerged as the dominant post-training paradigm for complex reasoning in LLMs, yet commonly suffer from policy entropy collapse during training. We conduct a first-order gradient analysis of token-level entropy dynamics under GRPO and identify a token-level credit assignment mismatch: the per-token entropy variation decomposes into the product of the trajectory-level advantage and an entropy sensitivity function over the next-token distribution, yielding an advantage-surprisal four-quadrant structure and a near-criticality property. Motivated by it, we propose STARE (Surprisal-guided Token-level Advantage Reweighting for policy Entropy stability), which identifies entropy-critical token subsets via batch-internal surprisal quantiles, selectively reweights their effective advantages, and incorporates a target-entropy closed-loop gate for stable entropy regulation. A","title":"STARE: Surprisal-Guided Token-Level Advantage Reweighting for Policy Entropy Stability","url":"https://arxiv.org/abs/2606.19236","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.19236v1 Announce Type: cross \nAbstract: Reinforcement Learning with Verifiable Rewards algorithms like GRPO have emerged as the dominant post-training paradigm for complex reasoning in LLMs, yet commonly suffer from policy entropy collapse during training. We conduct a first-order gradient analysis of token-level entropy dynamics under GRPO and identify a token-level credit assignment mismatch: the per-token entropy variation decomposes into the product of the trajectory-level advantage and an entropy sensitivity function over the next-token distribution, yielding an advantage-surprisal four-quadrant structure and a near-criticality property. Motivated by it, we propose STARE (Surprisal-guided Token-level Advantage Reweighting for policy Entropy stability), which identifies entropy-critical token subsets via batch-internal surprisal quantiles, selectively reweights their effective advantages, and incorporates a target-entropy closed-loop gate for stable entropy regulation. A","title":"STARE: Surprisal-Guided Token-Level Advantage Reweighting for Policy Entropy Stability","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-18T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.19236"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a72376fe342e980489693b3c3bff0373ef98b0653d04b8d34fd4883a8210cfb081e7582e244ec1037891b1c181ba955484f02b53c8236dd84ed9fed1ee017806","signer":"crovia.substrate","subject":{"observed_at":"2026-06-18T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.19236"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f4fe1f917497ce1c88c877398f887de86ba19eb3b5f65a3740003f389e82cfdb","leaf_index":233415,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"30b55655b8e7e69620d0840f28d62923536edf3c1596a048f6cd97b0e2baddb4","side":"left"},{"sibling":"d8fd9810903384abe5fd7ecf413e017f0237ec4a8a3a579b31d916a328068933","side":"left"},{"sibling":"ed049467ee3ce261a9419a32ca61a4fa2a60360469a94fc55bb463b97ceb48f7","side":"left"},{"sibling":"698d2ac0eb140605134089962209affd47aa8a7b34bd23082a71337a48730a23","side":"right"},{"sibling":"60dc775dd7c0d2c4ddfbb530920ebc933e906f33be0dd60a75f4cbb0766bc514","side":"right"},{"sibling":"8c610da56e97c6514b09dd5f45c229dc3a880e5ee127aa6c1cb747aa9abcccc0","side":"right"},{"sibling":"8942c0022e58c4389bb72bd28400c56919b0d25c91e8d4c9f09dec03626b6f4b","side":"left"},{"sibling":"b4c26795680b2096400acbdc34159290b2a0589778819fc7e052c7a60bb5c873","side":"left"},{"sibling":"429c2a92a65e6eeaa2eda0a35fdb9e541472a1eace4c69a4d01a618a659a110f","side":"left"},{"sibling":"d97d1ebe04af6ea572f9d4334004d01026acffa3da5be2883c7566513c76e2c0","side":"left"},{"sibling":"571eb56e7ce00fe1f38d0ac4fc56828d01b2cfc1ab9089cde245c0656bee0514","side":"left"},{"sibling":"7ac50038a8ced3aeaf1194a2407a4a09346b0e4399da36ecde4390675a0c4bf1","side":"left"},{"sibling":"8c5e2b48dc31ef0edcd35c3db048235aa78cc48443aa2a8da3aa6e9b5524d2c4","side":"right"},{"sibling":"e616c34dbaf9456d5a6d3e2da82cde8621293c9f6d8a4cf9e441d7fd9cc81579","side":"right"},{"sibling":"94c0c932e61657f5e37fdba43f6ca9eddea8359425a7c1558dabe566911d5304","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":234491,"merkle_root":"02576a6980e38bab47864ae2c57b5a5ff21e554e9bdf8f64bdf28155ff1aabec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260618T143732Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-18T18:33:39Z","sig_algorithm":"ed25519","signature":"b6c708778fc38b7789a2b91156cfe87252a7cd3a1d29121cba11a0c78f8cf104ca3019fc50a962fa5a216bcc4922fc8f3f69c04d1f8332c6dc0d931e1012e502","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1d268ca1f183a0a12148483a4d5dc954f69fe913cac241c41e3401795a290b24"}}