{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_34f703f99ee12e90689a18416ee288fef495734459d5f196e8071e3d5d334fe4","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_34f703f99ee12e90689a18416ee288fef495734459d5f196e8071e3d5d334fe4","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"66465422bc4b7471beaa46532ec8c29c132584a2f3664919e265289f772ec5b6","published":"Thu, 11 Jun 2026 00:00:00 -0400","receipt_hash":"66465422bc4b7471beaa46532ec8c29c132584a2f3664919e265289f772ec5b6","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"66465422bc4b7471beaa46532ec8c29c132584a2f3664919e265289f772ec5b6","observed_at":"2026-06-11T04:43:37.662146Z","parent_run_hash":"5267801b61ae0d882196b5f37208a9a1633905a64ca7d933f1fa5075cd861491","published":"Thu, 11 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.11499v1 Announce Type: cross \nAbstract: The performance of modern language models depends critically on pretraining data composition. Yet existing data selection methods rely on auxiliary classifiers for document scoring or mixture optimization, adding computational overhead and dependence on labeled data. We propose WebGraphMix, a lightweight data selection framework that computes structural centrality scores over the Common Crawl host-level web graph and uses them to vary the proportion of central versus peripheral documents in the pretraining mixture. We hypothesize that central hosts expose models to reusable abstractions, while peripheral hosts encode specialized, long-tail knowledge. WebGraphMix computes centrality scores efficiently at web scale, requiring no model training, labeled data, or downstream supervision. We integrate WebGraphMix into the DataComp-LM pipeline and train models at 400M and 1B parameter scales with 8B and 28B tokens respectively, evaluating on ","title":"Hubs or Fringes: Pretraining Data Selection via Web Graph Centrality","url":"https://arxiv.org/abs/2606.11499","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.11499v1 Announce Type: cross \nAbstract: The performance of modern language models depends critically on pretraining data composition. Yet existing data selection methods rely on auxiliary classifiers for document scoring or mixture optimization, adding computational overhead and dependence on labeled data. We propose WebGraphMix, a lightweight data selection framework that computes structural centrality scores over the Common Crawl host-level web graph and uses them to vary the proportion of central versus peripheral documents in the pretraining mixture. We hypothesize that central hosts expose models to reusable abstractions, while peripheral hosts encode specialized, long-tail knowledge. WebGraphMix computes centrality scores efficiently at web scale, requiring no model training, labeled data, or downstream supervision. We integrate WebGraphMix into the DataComp-LM pipeline and train models at 400M and 1B parameter scales with 8B and 28B tokens respectively, evaluating on ","title":"Hubs or Fringes: Pretraining Data Selection via Web Graph Centrality","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-11T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.11499"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4e21ac4ff89d6df23678719b237de24e406fbad0e938f706f1026806fd1686ed793943c4e561ce7c8b0ae535c8fb833b5f72da4791958d24877636a829c2c90e","signer":"crovia.substrate","subject":{"observed_at":"2026-06-11T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.11499"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"fccc94b9ea80065f41d82a378fc4e7ca45a81aa85101602ccdd7736c1e581fbd","leaf_index":227409,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c54e6aa64f9080fc49fc8f9fa2b8a285cc8bf496d1c548d78b102ab29f79706c","side":"left"},{"sibling":"cfe3aac34b91bb2da5d30d1685180e9303346c5f57f768d9e83fbc7167a0d3dd","side":"right"},{"sibling":"09b47a882d0fe66e2dbdbf64663d98892eb8c6c455aa0b80ced82ec581bc71fd","side":"right"},{"sibling":"30eea733b580958bb9fc30024af6af148fa6f0931c3961d81911258fb9ac34cf","side":"right"},{"sibling":"fb3402cb2beda4bc8df6aa24b46d730391c740d401464ba0de09e5c1c8891807","side":"left"},{"sibling":"93550d03a83a2be853ab89844eeeebad093f0a1cd104636967f7c5c3853d7974","side":"right"},{"sibling":"96bcd6ed4c2b2c36b81b61babb806f683bcc00f55d4beb48afc2ce7777850ac6","side":"left"},{"sibling":"a6d2c01df24bdf78d7a3a20470793a1585a197c5d8c3f07fc4049ed07458246d","side":"right"},{"sibling":"2cfac7f042209c8533c6031bddc4a83bc156e0595f1b28efefbda208904f338a","side":"right"},{"sibling":"04e399458c5b36988cae0bf1c6dbe1b01349003b15cb5aa43f95c55acffe4ec3","side":"right"},{"sibling":"1383228337d54218bd8e5563aebb0b0dfe15c5269e3d5138e64c261d6130a88b","side":"right"},{"sibling":"57cb49c192550231071a0bf53a0821da2f79c585ec6c8d0fc76cebd62ccd78b2","side":"left"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_34f703f99ee12e90689a18416ee288fef495734459d5f196e8071e3d5d334fe4"}}