{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b9369503a98960c9e7c2b0d2fb2f7b0de2e8acfceff893daaff7ba54684ee9ef","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b9369503a98960c9e7c2b0d2fb2f7b0de2e8acfceff893daaff7ba54684ee9ef","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"5b3845028b43564a4b7f14e3a3f477cf09cf355d594602f98fdb6bad9ef269be","published":"Fri, 17 Jul 2026 00:00:00 -0400","receipt_hash":"5b3845028b43564a4b7f14e3a3f477cf09cf355d594602f98fdb6bad9ef269be","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"5b3845028b43564a4b7f14e3a3f477cf09cf355d594602f98fdb6bad9ef269be","observed_at":"2026-07-17T04:43:38.280949Z","parent_run_hash":"113193614a8af99887180226d4e28a8b71d957da5fe3694f0e7a56807c145504","published":"Fri, 17 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.15232v1 Announce Type: cross \nAbstract: A tokenizer fixed at the start of pre-training allocates vocabulary in proportion to the pre-training corpus, reflecting the deployment priorities at that time. When those priorities shift, languages added later are split into many more tokens per word, which can raise latency, compute, and energy consumption for users of those languages. Cloud models can afford a broad vocabulary because the embedding and LM-head matrices are a small fraction of their parameters. On a compact model those matrices are a material share of per-token decode bandwidth, so on-device models ship small vocabularies and accept fragmentation outside a fixed language set. We present tokenizer expansion, an in-place recipe for upgrading a pre-trained model's tokenizer when the model producer controls its design. We continue the existing tokenizer's BPE merges on a multilingual corpus, so most source tokens carry over unchanged as single tokens and every new token","title":"In-Place Tokenizer Expansion for Pre-trained LLMs","url":"https://arxiv.org/abs/2607.15232","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.15232v1 Announce Type: cross \nAbstract: A tokenizer fixed at the start of pre-training allocates vocabulary in proportion to the pre-training corpus, reflecting the deployment priorities at that time. When those priorities shift, languages added later are split into many more tokens per word, which can raise latency, compute, and energy consumption for users of those languages. Cloud models can afford a broad vocabulary because the embedding and LM-head matrices are a small fraction of their parameters. On a compact model those matrices are a material share of per-token decode bandwidth, so on-device models ship small vocabularies and accept fragmentation outside a fixed language set. We present tokenizer expansion, an in-place recipe for upgrading a pre-trained model's tokenizer when the model producer controls its design. We continue the existing tokenizer's BPE merges on a multilingual corpus, so most source tokens carry over unchanged as single tokens and every new token","title":"In-Place Tokenizer Expansion for Pre-trained LLMs","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-17T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.15232"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1521cb0c7e57a7318b42954911307738783277d55068e359126160d8dabf82bd7a643b5c1c629f82fe826c54d7143c3770c07a523f8a0b2290e1865d5b4dcc0d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-17T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.15232"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"4902b22c387e161613324a9b0539eb5f90459c8c0c728031e980931294513885","leaf_index":323172,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6b2b84d2d61405f4ca3e01d8c4db2938298412fe5bce51f808b4a523fd211922","side":"right"},{"sibling":"b5d9b5db18ed15f64c31754b785e526437dde96c10a79555c9fdac76b163cf1c","side":"right"},{"sibling":"90b62893b456261e39bd53cf4678497c5e74bb8399e86a659cb5864133537df5","side":"left"},{"sibling":"93050a3b0d1b5ab235f5c1a7b39f012867ab9fa9fb0c12af1b4af3a605a1025b","side":"right"},{"sibling":"cd79d37482528840323d33289a3b71a070acd02cc018ab1271dba87e1b0ad578","side":"right"},{"sibling":"00bd52f3534c19833a0f638a09d0c0670180e59cb7ba9a2aeb3b4c000ac66696","side":"left"},{"sibling":"65f8a681723598b39869e2bf9640e4e77ef6eea44f85f6f14b4e77bf40cd5f40","side":"left"},{"sibling":"de6ef12c0ebc7e3e48430c64265163103eb92448efd698dbc2b104ada5a5389d","side":"right"},{"sibling":"4242cc570ec8c36a37f3f6f20dcae20b49fccc17c7b53ba47d8715e465bec585","side":"right"},{"sibling":"be025f48721bfc0ca7107f0454bda3ab460e50539f0caeb1bf839a8dabcf036c","side":"left"},{"sibling":"05a09763743cdc09fc45cf454e4e3ea4a0d1cd74f9c8162b2a57e2c873160908","side":"left"},{"sibling":"de3120ef2488b8a791a686b47257da4e612256abdfcdda7519265e7edd47d041","side":"left"},{"sibling":"e86f56a4883492da5b5e7b0201324c52946e865e69b99ebb532f41fe3c658ee4","side":"right"},{"sibling":"34d85f6ad6cc7dfa79d90e2b9ff99a561bcdc75b0301bbbd3e83861f54535c1e","side":"left"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"9b11714124b9b951ff9450b0ee9d625a0b70b6da2cf388bd3df1475eec0b17ba","side":"right"},{"sibling":"a4523a9014d45df43e006e9210a73428c380d771f2c650a1b986910759b0cdf7","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":323382,"merkle_root":"2f4d32419c80a9600aba5a480fc3fb7012ec0a695c91a1b055048e78760b65ca","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260717T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-17T05:38:31Z","sig_algorithm":"ed25519","signature":"495308c7bf004117807331d3f71d0b079f6bd7ed7737faa58c773b1b3e80ee84928d2d8519cc7d2b809506501aada1be6546f72c7ecda3dad5445cbb44502209","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b9369503a98960c9e7c2b0d2fb2f7b0de2e8acfceff893daaff7ba54684ee9ef"}}