{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_86dc5da9b4762857b8af470106a845f4cf788cc12ff8069965ba5cd4f981d1b3","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_86dc5da9b4762857b8af470106a845f4cf788cc12ff8069965ba5cd4f981d1b3","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"88dadaab793a4b3c30c483493245e13667004d8f55497be91341704a8f52435d","published":"Tue, 19 May 2026 00:00:00 -0400","receipt_hash":"88dadaab793a4b3c30c483493245e13667004d8f55497be91341704a8f52435d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"88dadaab793a4b3c30c483493245e13667004d8f55497be91341704a8f52435d","observed_at":"2026-05-19T04:43:36.782648Z","parent_run_hash":"fefa4c726316a95c5dda9fc1ca07a38a811cf7ffa9825b2f09b365abacd9b32d","published":"Tue, 19 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.17379v1 Announce Type: cross \nAbstract: Large language models pretrained on general-domain corpora often exhibit tokenization inefficiencies when applied to specialized domains. Although continual pretraining for domain adaptation partially alleviate performance degradation, it does not resolve the fundamental vocabulary mismatch. To address this gap, we introduce a targeted parameter-efficient domain adaptation approach that combines vocabulary adaptation with pretraining for LLM-based text summarization. Our unified framework augments pretrained tokenizers with domain-specific tokens while selectively replacing under-trained and unreachable tokens to limit parameter growth. We evaluate our approach on Llama-3.1-8B and Qwen2.5-7B across legal and medical summarization tasks on a challenge-oriented evaluation protocol focused on expert-driven text and summaries which typically has higher concentration of over-fragmented Out-of-Vocabulary (OOV) words. The vocabulary adaptatio","title":"Learning Faster with Better Tokens: Parameter-Efficient Vocabulary Adaptation for Specialized Text Summarization","url":"https://arxiv.org/abs/2605.17379","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.17379v1 Announce Type: cross \nAbstract: Large language models pretrained on general-domain corpora often exhibit tokenization inefficiencies when applied to specialized domains. Although continual pretraining for domain adaptation partially alleviate performance degradation, it does not resolve the fundamental vocabulary mismatch. To address this gap, we introduce a targeted parameter-efficient domain adaptation approach that combines vocabulary adaptation with pretraining for LLM-based text summarization. Our unified framework augments pretrained tokenizers with domain-specific tokens while selectively replacing under-trained and unreachable tokens to limit parameter growth. We evaluate our approach on Llama-3.1-8B and Qwen2.5-7B across legal and medical summarization tasks on a challenge-oriented evaluation protocol focused on expert-driven text and summaries which typically has higher concentration of over-fragmented Out-of-Vocabulary (OOV) words. The vocabulary adaptatio","title":"Learning Faster with Better Tokens: Parameter-Efficient Vocabulary Adaptation for Specialized Text Summarization","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-19T04:43:36Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.17379"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:87f31fa9a49f9b1fd8c20558c68fd29aeca22be76b1bd1ddda973f7a258c3e5c407f23d120624179dd6d51bcd104febd4ccb7906e515130b08a8b68c884e6c0d","signer":"crovia.substrate","subject":{"observed_at":"2026-05-19T04:43:36Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.17379"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"e3ab7ed9ec51283b58e8c9e2e3165136c0d6012e75cb489c3fc2c5dde2373193","leaf_index":142734,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f7045743232cf50e6740642c51d0fe3eb69fe1624177e14977dc5d30a1579d01","side":"right"},{"sibling":"31fc2976b329292a2e6998aa0331367d71114ea9c23fd87b0f6e6cb3c36b97ee","side":"left"},{"sibling":"d36ca7807f5e42002e688fbf8895252b57c655550ff58c24b534d21f7b5c39fa","side":"left"},{"sibling":"fd5577509727bf508a974adcd1748711fbabd257a3e7558c0eb0404e33be6aa2","side":"left"},{"sibling":"2ed796190835ef5598ff7b1b12da2634525ccf2d19801ffc27bb1d2f9c0f0b47","side":"right"},{"sibling":"5b96c198dbef2b0959c887616bb5b175adf0309d23c263eef954f42dd105718e","side":"right"},{"sibling":"fc791956c7f3340d3c458603b3d8eb2cf5b9c49d863090e0c16aa81ddd04a07e","side":"right"},{"sibling":"bda8ed627eb0c8b21066bfa3a60b10b3fd1daf60f49295408a27bab24b140eb3","side":"left"},{"sibling":"f1c796d3bd453570426dcee9ab20072f19762202f84b7993afba2c7c0ee9ec3b","side":"left"},{"sibling":"2e0ce989d789c88e796991ef014ce7e4e1f96c0d4ede9da8d31afcf5ba6a8f46","side":"right"},{"sibling":"202f1bead178ef3785968d50d3d188264a95192a077654c331612e04a34cbfbe","side":"left"},{"sibling":"72249c8c8b068386e35d16f4bd0bbeb9ba820ca217ef0f0d28396c9fe493f5f0","side":"left"},{"sibling":"ea64599340f7ffdf17ad0cbc1d9401ef8870a347e3847bdc106d06b1673df09c","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"4db1f363729507e27a60851cf6ed334d7b9acdef194ed7d419aba4d2bd367a4a","side":"right"},{"sibling":"a86ee18c45e7fcc408b6007eaece05aa75b2d9ae30252e9e878462b4dffbef7b","side":"right"},{"sibling":"1d18e7663d43ccff0122ecc7ee12645bb16afb607b218e81b1ea2408f863cb78","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":143302,"merkle_root":"999156d40a7c61d9ddd52b7338f3cbda3e68f53bace070c7b616ea194e23b123","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260519T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-19T05:37:30Z","sig_algorithm":"ed25519","signature":"b1a252cc66ff32bed1d10dd88a6b2a200e3856d3dbcfcc4ee55e02e00f3d548e854ed9c544704b222bd5d315492c4a935ba2d90d727c585a67899b0ad602fc05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_86dc5da9b4762857b8af470106a845f4cf788cc12ff8069965ba5cd4f981d1b3"}}