{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_57857e6564724cb4e4bed99c04a483a0313930690bfd5db14604a8cd9f6e3048","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_57857e6564724cb4e4bed99c04a483a0313930690bfd5db14604a8cd9f6e3048","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"d8be08fc976a3fa1a9c15fb0116366fe421df78f64233d5629efa8fc9062fd74","published":"Mon, 08 Jun 2026 00:00:00 -0400","receipt_hash":"d8be08fc976a3fa1a9c15fb0116366fe421df78f64233d5629efa8fc9062fd74","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"d8be08fc976a3fa1a9c15fb0116366fe421df78f64233d5629efa8fc9062fd74","observed_at":"2026-06-08T04:44:02.392073Z","parent_run_hash":"4b9e67a023632e16a32d228bb97fee209911f388e0a8dbf20b5a4ec02729c20f","published":"Mon, 08 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.07001v1 Announce Type: cross \nAbstract: High-quality training data is essential to large language models (LLMs) and typically requires extensive and costly manual curation. Existing automatic data preparation methods rely on predefined pipelines or customized human instructions, which limits their adaptability to diverse data distributions and lacks principled guidance from high-quality examples. In this paper, we introduce DataEvolver, the first self-evolving data preparation system that automatically constructs pipelines to transform raw data into high-quality data. DataEvolver employs a multi-level mechanism to ensure both pipeline executability and effectiveness. At the operator level, it incrementally expands the operator set to construct a logical plan while resolving dependency conflicts. At the pipeline level, it instantiates logical plans into executable code and iteratively refines pipeline orchestration through a feedback loop that reduces the distribution gap bet","title":"DataEvolver: Automatic Data Preparation for Large Language Models through Multi-Level Self-Evolving","url":"https://arxiv.org/abs/2606.07001","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.07001v1 Announce Type: cross \nAbstract: High-quality training data is essential to large language models (LLMs) and typically requires extensive and costly manual curation. Existing automatic data preparation methods rely on predefined pipelines or customized human instructions, which limits their adaptability to diverse data distributions and lacks principled guidance from high-quality examples. In this paper, we introduce DataEvolver, the first self-evolving data preparation system that automatically constructs pipelines to transform raw data into high-quality data. DataEvolver employs a multi-level mechanism to ensure both pipeline executability and effectiveness. At the operator level, it incrementally expands the operator set to construct a logical plan while resolving dependency conflicts. At the pipeline level, it instantiates logical plans into executable code and iteratively refines pipeline orchestration through a feedback loop that reduces the distribution gap bet","title":"DataEvolver: Automatic Data Preparation for Large Language Models through Multi-Level Self-Evolving","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-08T04:44:02Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.07001"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:9d4e6f19657b170aea22f801e0837f5f85d20252ae0eac6e0ec0fb880fd35c74f2df1ba2db0b44c5fd6ea9170ce200d8a3e9ef3752b222f376e40c8144f30806","signer":"crovia.substrate","subject":{"observed_at":"2026-06-08T04:44:02Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.07001"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"a534fac528db851346d0530394c6d3a0e84822d741b20b06c96de42226021ebb","leaf_index":223748,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"88186ccd7e2b008606ece73d4fd8aca260f06a9e000bc1335a696ff6dee30acf","side":"right"},{"sibling":"e944238c83a9b9a4b0b81db71732dd3804b4e20098d2affaee1d508e036b4cb5","side":"right"},{"sibling":"15da7c099a1458905760d162df951421811918fb597e3bbb48ed4e12045d1378","side":"left"},{"sibling":"cd289855ea3ed1068e1e7f2260cf3a392d920a53a3397cffd97405a11bcfd2db","side":"right"},{"sibling":"f539768a92d987da9f78653462c259b5a8372341d0cc05dc64889e7facb8989d","side":"right"},{"sibling":"b7b453ef2baf0d34e6ebbd1fa44cad77319e644524315b1f13ce29253a2c9015","side":"right"},{"sibling":"4e304d5d7910cf045b4d600075b2e1b6a8e413ccb358cfc3bad788ceebcc2de9","side":"right"},{"sibling":"4fcf914698d33c1c26df7cb3719b9425a51cc354d8550fca9c3f04fdd2314f2a","side":"right"},{"sibling":"fad4d9627e4b025e7840b4e896082fb8291cbf1a4c05653b3a789c8a4205d4b5","side":"right"},{"sibling":"e8dcea313a54920d83e4f72d5a4223f986c719f264241d171c4712efaaf1fc54","side":"left"},{"sibling":"5480e1ea31f4744f9bd7c4261771fe51f2cdb01e705cc17320bfc202d935ca12","side":"right"},{"sibling":"24fdc29d461691aedb6fa920758206b5bb43851f477ef7a04c34aaed84b8971b","side":"left"},{"sibling":"036922da4e1e2c46d948f070454bfad299b7406fb00735ea9d8bd1e687f5f445","side":"right"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"87c6b850dfec08ac35a693d9db3a3315250a68adb1cfab9b1015f212b63b15bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":224761,"merkle_root":"e9f7b49b652e869ab97ffba9c5a31356b2d0e3dc5d00bb28944adf737c46b1e7","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260609T103805Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-09T14:15:34Z","sig_algorithm":"ed25519","signature":"8ad8076fb12c8e486ae1d1559a9a7ba8e2ee996a9ad3d8ba7bcdbdbd88ab3a15bcb429707aca6d3e9d8b97e2ba755b3dcc77b1abb6601ccb829842719a6fb30d","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_57857e6564724cb4e4bed99c04a483a0313930690bfd5db14604a8cd9f6e3048"}}