{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_540ff4602b4e7fa98c8c77bab64960a6c8a66c0c2d318272a51037c6b962ac83","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_540ff4602b4e7fa98c8c77bab64960a6c8a66c0c2d318272a51037c6b962ac83","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"92d9c95dddb9bccfa98bc4486ea09126f8c0387be986df1d2648da87d030c2c2","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"92d9c95dddb9bccfa98bc4486ea09126f8c0387be986df1d2648da87d030c2c2","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"92d9c95dddb9bccfa98bc4486ea09126f8c0387be986df1d2648da87d030c2c2","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.20089v1 Announce Type: cross \nAbstract: Persian pretrained language models (PLMs) are still limited by the scarcity of large-scale, high-quality pretraining corpora and by insufficient evaluation beyond standard classification and NER tasks. We present IHUBERT, a monolingual Persian PLM trained from scratch with the RoBERTa-base encoder (125M parameters) on a 45 GB curated subset of the Sepahr-Danesh collection (about 7-8B tokens). To improve corpus quality and reduce redundancy, we employ a multi-stage preprocessing pipeline that includes normalization, exact and near-duplicate removal, anonymization, and vector-database-based semantic deduplication for distribution balancing control across domains and registers. We additionally train a 139k-vocabulary BPE tokenizer on the full pretraining corpus to better capture Persian morphology and orthographic variation. IHUBERT is evaluated on seven Persian NLU benchmarks covering NER, sentiment analysis, topic classification, NLI, e","title":"IHUBERT: Vector-Based Semantic Deduplication and Domain-Balanced Pretraining for Persian Resources","url":"https://arxiv.org/abs/2606.20089","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.20089v1 Announce Type: cross \nAbstract: Persian pretrained language models (PLMs) are still limited by the scarcity of large-scale, high-quality pretraining corpora and by insufficient evaluation beyond standard classification and NER tasks. We present IHUBERT, a monolingual Persian PLM trained from scratch with the RoBERTa-base encoder (125M parameters) on a 45 GB curated subset of the Sepahr-Danesh collection (about 7-8B tokens). To improve corpus quality and reduce redundancy, we employ a multi-stage preprocessing pipeline that includes normalization, exact and near-duplicate removal, anonymization, and vector-database-based semantic deduplication for distribution balancing control across domains and registers. We additionally train a 139k-vocabulary BPE tokenizer on the full pretraining corpus to better capture Persian morphology and orthographic variation. IHUBERT is evaluated on seven Persian NLU benchmarks covering NER, sentiment analysis, topic classification, NLI, e","title":"IHUBERT: Vector-Based Semantic Deduplication and Domain-Balanced Pretraining for Persian Resources","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.20089"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4fb707366ae82bbc1191e42b712c025547fb9ae6c8bb14308e7e11fada3dc7d0c89cc733bd18d2339e844f17aabc8e141fdf070642ef9c84510b6c6da8d35603","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.20089"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"cf3160799768df5414b8f9369c1e30afeb29f9670898a4945c4339820fe24f6e","leaf_index":235638,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"94499f9473395269967a09529ecaf4d3f941191a03f233ed888e9c51caf55a68","side":"right"},{"sibling":"4844ea86dd98721d5c44d9182e939321d97fae3d51d77db8cf37ed2cc951caef","side":"left"},{"sibling":"1115f1e2750967b002eb7fdf00a24cd225e166b1a9978679393e4b26ae79877c","side":"left"},{"sibling":"40bdf776aa2437f1cfb4ef1326ce478c34a7921c41e0a6451d7b911b49732962","side":"right"},{"sibling":"dde503dbed1a57b88af2e8da9ed531152ff23c391e0f52daa3c1097ca199df6e","side":"left"},{"sibling":"6f166516bd3a54caabf9ea50438bb1770d8367b867a8338a79ec80c148924d4b","side":"left"},{"sibling":"eb70d4eaf5e9558deed789daa5addfb31fc38e81a3f06b590069834b28d61e1c","side":"left"},{"sibling":"c0ab0c7dd98a2c17ee965cb19831d49d3a82a2ba1812d8a677f78274ab6f7c98","side":"right"},{"sibling":"aa68ebe8f5e8e96388fc8d1af3aa08be7ccd27ab4cebcf5913560c48d877bc27","side":"right"},{"sibling":"8253d44cf1ed30d3ab19c2b339fb4000a1fa173182c65390e9e8dabf8173b9e9","side":"right"},{"sibling":"e2bf9b60400244c698c0196109f54323457abc5b64dee08ec33ab14cc4faaef7","side":"right"},{"sibling":"86664e7f68ba08b8dfcf77dda51a4dfa7fcfc986d4ad7c704ffb71b669202da7","side":"left"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_540ff4602b4e7fa98c8c77bab64960a6c8a66c0c2d318272a51037c6b962ac83"}}