{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_9f43dff19b0e148f56925a7a572667bd6fc5531bc46e83545c3a6987993b0e31","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_9f43dff19b0e148f56925a7a572667bd6fc5531bc46e83545c3a6987993b0e31","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a8b490729559d4c00fd4fae2e2198d9ce9b6206197a9f0c69ac0993b9dac0c77","published":"Tue, 09 Jun 2026 00:00:00 -0400","receipt_hash":"a8b490729559d4c00fd4fae2e2198d9ce9b6206197a9f0c69ac0993b9dac0c77","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a8b490729559d4c00fd4fae2e2198d9ce9b6206197a9f0c69ac0993b9dac0c77","observed_at":"2026-06-09T04:43:45.619596Z","parent_run_hash":"f2344865fd128464efd1bacba326b5a7ccea707694b8c5650dd51ae8c46ac8a1","published":"Tue, 09 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.07996v1 Announce Type: cross \nAbstract: Pretraining is fundamental to the development of Large Language Models (LLMs), yet the opacity of pretraining data complicates model analysis and raises ethical, legal, and fairness concerns. Detecting whether specific datasets were used during pretraining is, therefore, critical. Existing state-of-the-art methods typically rely on access to model probability distributions, making them unsuitable for closed-source LLMs that provide only input-output interfaces. To address this limitation, we introduce Masked Corpus-level Pretraining Data Detection (MC-PDD), a novel method inspired by the masked language modeling paradigm. MC-PDD masks highly specific tokens in each text and prompts the LLM to predict the missing content. It then assesses whether the difference in prediction hit rates between a candidate corpus and a reference non-member corpus is statistically significant. Based on this comparison, MC-PDD determines whether the candida","title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","url":"https://arxiv.org/abs/2606.07996","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.07996v1 Announce Type: cross \nAbstract: Pretraining is fundamental to the development of Large Language Models (LLMs), yet the opacity of pretraining data complicates model analysis and raises ethical, legal, and fairness concerns. Detecting whether specific datasets were used during pretraining is, therefore, critical. Existing state-of-the-art methods typically rely on access to model probability distributions, making them unsuitable for closed-source LLMs that provide only input-output interfaces. To address this limitation, we introduce Masked Corpus-level Pretraining Data Detection (MC-PDD), a novel method inspired by the masked language modeling paradigm. MC-PDD masks highly specific tokens in each text and prompts the LLM to predict the missing content. It then assesses whether the difference in prediction hit rates between a candidate corpus and a reference non-member corpus is statistically significant. Based on this comparison, MC-PDD determines whether the candida","title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-09T04:43:45Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.07996"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:0d0e05aa01a7ef187b651afb7c297824607eca0ba90c5ad69ba15eb9e64dcc1338404537325aa9db6514c0876b40c88fea6e1dd8de9e838df789147a0ed43100","signer":"crovia.substrate","subject":{"observed_at":"2026-06-09T04:43:45Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.07996"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"88277592781f103ef53ff347e5002d0c17d3cb363851b52b88384be28c2de6ac","leaf_index":224223,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6fdb4931683b609aa8458bd17a571a97e504c3b8da64acb5e92565b3d392d513","side":"left"},{"sibling":"594c028319ca8b30d9ffc84b1b3c03d67606c7329a5891ef53b70cbc0544049d","side":"left"},{"sibling":"1a4550baacb4c532ca08dacad32bb72a867d2c63f019e0b9c740265eab4e2379","side":"left"},{"sibling":"4888c2f4641079e46e7045fc54351ed5ecaf59660a42c395b64e20286340a32d","side":"left"},{"sibling":"3d77e6ac75a3f723574457d5d21fa73e1b530caf2e5fb8f5e0551fab6947aabc","side":"left"},{"sibling":"4610d12e19064d346205809d9780399175b7ebcd09d065cb6a9dfdea3b6d69c4","side":"right"},{"sibling":"a3c160f5d9575cb67b5a4d2e5ed67ba2b93e251f889a7eb78c6edde205f0f4e5","side":"left"},{"sibling":"10fa631530dc46d85fa0d114403a05652fa3690c3219b57f26650e0db77a8036","side":"left"},{"sibling":"dd5d61819b080d26c4380e26887a8eb889d51e2c022f796a39855defe6f68d95","side":"left"},{"sibling":"e8dcea313a54920d83e4f72d5a4223f986c719f264241d171c4712efaaf1fc54","side":"left"},{"sibling":"5480e1ea31f4744f9bd7c4261771fe51f2cdb01e705cc17320bfc202d935ca12","side":"right"},{"sibling":"24fdc29d461691aedb6fa920758206b5bb43851f477ef7a04c34aaed84b8971b","side":"left"},{"sibling":"036922da4e1e2c46d948f070454bfad299b7406fb00735ea9d8bd1e687f5f445","side":"right"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"87c6b850dfec08ac35a693d9db3a3315250a68adb1cfab9b1015f212b63b15bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":224761,"merkle_root":"e9f7b49b652e869ab97ffba9c5a31356b2d0e3dc5d00bb28944adf737c46b1e7","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260609T103805Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-09T14:15:34Z","sig_algorithm":"ed25519","signature":"8ad8076fb12c8e486ae1d1559a9a7ba8e2ee996a9ad3d8ba7bcdbdbd88ab3a15bcb429707aca6d3e9d8b97e2ba755b3dcc77b1abb6601ccb829842719a6fb30d","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_9f43dff19b0e148f56925a7a572667bd6fc5531bc46e83545c3a6987993b0e31"}}