{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_49ef04798ff4d0f3d4035a51a7bcdf271933294bf5b4aa09e53cb911e2993780","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_49ef04798ff4d0f3d4035a51a7bcdf271933294bf5b4aa09e53cb911e2993780","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a02ac537c7ce0f8e1af4590917697109b1f874c8a0e89975c074061182c3ebec","published":"Tue, 16 Jun 2026 00:00:00 -0400","receipt_hash":"a02ac537c7ce0f8e1af4590917697109b1f874c8a0e89975c074061182c3ebec","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a02ac537c7ce0f8e1af4590917697109b1f874c8a0e89975c074061182c3ebec","observed_at":"2026-06-16T04:43:43.281320Z","parent_run_hash":"eb6edcf82c3507c59161a4ab46d2e904e507004f44677402bb24d106997ed7c2","published":"Tue, 16 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2506.16738v2 Announce Type: replace-cross \nAbstract: With the rapid progress of speech language models (SLMs), discrete speech tokens have emerged as a core interface between speech and text, enabling unified modeling across modalities. Recent speech tokenization approaches aim to isolate semantic information from low-level acoustics to better align with language models (LMs). In particular, previous methods use self-supervised learning (SSL) teachers such as HuBERT to extract semantic representations, which are then distilled into a semantic quantizer to suppress acoustic redundancy as well as capture content-related latent structures. However, these tokenizers often operate at relatively high frame rates, producing token sequences significantly longer than their textual counterparts and hindering seamless integration with pretrained LMs. Although recent methods attempt to reduce the token rate by applying uniform average pooling to SSL features, this can over-smooth content-bea","title":"LM-SPT: LM-Aligned Semantic Distillation for Speech Tokenization","url":"https://arxiv.org/abs/2506.16738","vendor":"arxiv_cs_ai"},"summary":"arXiv:2506.16738v2 Announce Type: replace-cross \nAbstract: With the rapid progress of speech language models (SLMs), discrete speech tokens have emerged as a core interface between speech and text, enabling unified modeling across modalities. Recent speech tokenization approaches aim to isolate semantic information from low-level acoustics to better align with language models (LMs). In particular, previous methods use self-supervised learning (SSL) teachers such as HuBERT to extract semantic representations, which are then distilled into a semantic quantizer to suppress acoustic redundancy as well as capture content-related latent structures. However, these tokenizers often operate at relatively high frame rates, producing token sequences significantly longer than their textual counterparts and hindering seamless integration with pretrained LMs. Although recent methods attempt to reduce the token rate by applying uniform average pooling to SSL features, this can over-smooth content-bea","title":"LM-SPT: LM-Aligned Semantic Distillation for Speech Tokenization","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-16T04:43:43Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2506.16738"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:bbd137193e7ce989d776f85ba9125e99b2432556bad8bef88eb04919dd594f1417c7f6db221f822a93001e37bcd9175c16b386042dea414758518974d264f509","signer":"crovia.substrate","subject":{"observed_at":"2026-06-16T04:43:43Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2506.16738"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7e30ff4cb092ecd3327daf3670ae1bbb35c58f59c6e37867db1d77b59a66464a","leaf_index":230700,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6c0460f90931cca73c413e368bf12334cbb9d915f2c99f594722a250a2fae11d","side":"right"},{"sibling":"246cdeeba54c250de54cb757a641e1711d65a554286e14cf69afff004b51369e","side":"right"},{"sibling":"5e78a4138615cc5ebf67105733bf3320afdbc44ca9654e4ff8a400118d3891f5","side":"left"},{"sibling":"6e68b87ba4423cf52546208ad30e923dcab38d36d3dac92e32299976363f06c3","side":"left"},{"sibling":"638a582b8b587425f71a31e3e28fd61953b1be5e8cbaf071fd32fc398692f804","side":"right"},{"sibling":"6c5c8405cf520ca718b3c423ff768b18d358e6a3ab0ddcb0b008efc6508d7fc6","side":"left"},{"sibling":"427fbf2f7401da34e38f276c49611c2b1cf6e816c0a116d36f0e04e54f7562d3","side":"right"},{"sibling":"7a295c86e2ca5719bf2e33d1fcc4bc631f05c6f31aeb907113e2c58401e3087c","side":"right"},{"sibling":"ee59602dad0bf74c97a32a93f0a1a19e7a12f2800791988a6fdf35611febe031","side":"left"},{"sibling":"0c5669692381d605223c74b8d30f70cd308e77e33d5e40ea84bb7b4f84f2d4d9","side":"right"},{"sibling":"d5b9f8b1a2c9f6a46e17982dfbe6ce1f3b5fa4e730220397f2253d114dcc8486","side":"left"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_49ef04798ff4d0f3d4035a51a7bcdf271933294bf5b4aa09e53cb911e2993780"}}