{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_427ed4697684be12b5bad8c1f9f1d0b92f868741b83cec7a3e7686b4bede68bc","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_427ed4697684be12b5bad8c1f9f1d0b92f868741b83cec7a3e7686b4bede68bc","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"96ee41671f5325816774d82728332dd526f611514e34ee0cad8d0fc12eb00790","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"96ee41671f5325816774d82728332dd526f611514e34ee0cad8d0fc12eb00790","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"96ee41671f5325816774d82728332dd526f611514e34ee0cad8d0fc12eb00790","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.19771v1 Announce Type: new \nAbstract: Reinforcement Learning with Verifiable Rewards (RLVR) has significantly advanced Large Language Model (LLM) reasoning; however, it faces a fundamental optimization instability: uniform token updates precipitate entropy collapse, leading to premature convergence to suboptimal strategies, whereas excessive Shannon Entropy maximization can cause entropy explosion, driving blind exploration toward incoherent reasoning chains. To resolve this dichotomy, we introduce the Independent Combinatorial Tokens (ICT) framework, which shifts the optimization focus from scalar uncertainty to the distributional properties of token logits. By leveraging the Jensen-Shannon (JS) divergence between token logits distributions, ICT identifies tokens with distinctive distributional patterns as critical branching points for guiding effective exploration in LLM reasoning. Our theoretical analysis, grounded in both Shannon and second-order R\\'enyi entropy, proves ","title":"Beyond Entropy: Learning from Token-Level Distributional Deviations for LLM Reasoning","url":"https://arxiv.org/abs/2606.19771","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.19771v1 Announce Type: new \nAbstract: Reinforcement Learning with Verifiable Rewards (RLVR) has significantly advanced Large Language Model (LLM) reasoning; however, it faces a fundamental optimization instability: uniform token updates precipitate entropy collapse, leading to premature convergence to suboptimal strategies, whereas excessive Shannon Entropy maximization can cause entropy explosion, driving blind exploration toward incoherent reasoning chains. To resolve this dichotomy, we introduce the Independent Combinatorial Tokens (ICT) framework, which shifts the optimization focus from scalar uncertainty to the distributional properties of token logits. By leveraging the Jensen-Shannon (JS) divergence between token logits distributions, ICT identifies tokens with distinctive distributional patterns as critical branching points for guiding effective exploration in LLM reasoning. Our theoretical analysis, grounded in both Shannon and second-order R\\'enyi entropy, proves ","title":"Beyond Entropy: Learning from Token-Level Distributional Deviations for LLM Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.19771"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:531ec2b448129c145c552a4f082ce1645a5610a4a879d09ee851414f077f16a1056257fa3788558bf91200cc75ea7f7a55c85d547d8fded7cf081dd10486b30d","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.19771"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"5a7be9427a35ddfdf53fc84b976a5c14446730f367faa414aef9cb53cf87b551","leaf_index":235481,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"5541b7413c1fc8c2d5369ff057e4dfb65eba744fa5097671eba752e985c1efd6","side":"left"},{"sibling":"ea130c1123562cf658a940f39d3941082668483a9518653e77b8bb1079ddfa89","side":"right"},{"sibling":"ddc820789ee232d92a8367ece062e346c2a57702949abe23a230fb68d4a8fa14","side":"right"},{"sibling":"c9614e6ccc59e273dd7ee33a0abad3ae58c9a43b26b71a2e580fb017eb9f4193","side":"left"},{"sibling":"2508e3a114bb5ba4b2103300fcba8bdadc88708dcf642db0fcf3d64968d4d072","side":"left"},{"sibling":"bcc64a96b2351ef6571992656cd86c436090f98a58edd222113a9eab3bbe6e32","side":"right"},{"sibling":"696ac6bcf68966ee959b35539f0bed60b3216ccb53620defd29feb8dbcb2ac30","side":"left"},{"sibling":"95ba45f2fea2d822bc863962b7e478ea50086827d90fab88c0da065d454d7ac5","side":"left"},{"sibling":"00f27f149ddd2eb0d2cf60eaed9e1d55c662cf1cf69c95fbea03bba9d35f90c7","side":"left"},{"sibling":"797e0feb6bf826a55956c876711cd24824818c5a84a591f8cc06b95577b3405d","side":"left"},{"sibling":"f049d6e86b410f6f63921a6e3aa684efffa398fd23eed28245c43b156d404c5f","side":"left"},{"sibling":"9ec7f4f84de7e2057a02ac55686567b21acfd5beaecc7a84e9e48f3db17296c2","side":"right"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_427ed4697684be12b5bad8c1f9f1d0b92f868741b83cec7a3e7686b4bede68bc"}}