{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_dc7075965e28a76b38ee430fdd71ff84a38c924c914f8ee52577a6b6e4c5aca0","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_dc7075965e28a76b38ee430fdd71ff84a38c924c914f8ee52577a6b6e4c5aca0","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f20df2bc496958ccad7d653619a19c98e41febb3a0d58d68ddadbd8fa0a3274c","published":"Thu, 25 Jun 2026 00:00:00 -0400","receipt_hash":"f20df2bc496958ccad7d653619a19c98e41febb3a0d58d68ddadbd8fa0a3274c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f20df2bc496958ccad7d653619a19c98e41febb3a0d58d68ddadbd8fa0a3274c","observed_at":"2026-06-25T04:43:18.326769Z","parent_run_hash":"e8910cfae99fbdf5440a4f98691981c67edafcd7a0c47d48b244550cbfb5df92","published":"Thu, 25 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.24679v2 Announce Type: cross \nAbstract: Data preparation pipelines improve data quality in machine learning by transforming raw tables into learning-ready data through sequential cleaning and feature transformation operators. However, automatically constructing such pipelines is computationally difficult because operator sequences are combinatorial and end-to-end evaluation is expensive. Existing state-of-the-art (SOTA) Multi-DQN methods still face three key limitations: decoupled value estimators weaken long-horizon credit assignment, dataset context is only weakly injected into the policy, and exploration is inefficient in a sparse search space with many invalid states. To address these issues, we propose FlowPipe, a unified framework that formulates pipeline synthesis as conditional probabilistic flow generation over a directed acyclic graph. FlowPipe uses Conditional Generative Flow Networks (C-GFlowNets) with a Trajectory Balance objective to connect terminal validation","title":"FlowPipe: LLM-Enhanced Conditional Generative Flow Networks for Data Preparation Pipeline Construction","url":"https://arxiv.org/abs/2606.24679","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.24679v2 Announce Type: cross \nAbstract: Data preparation pipelines improve data quality in machine learning by transforming raw tables into learning-ready data through sequential cleaning and feature transformation operators. However, automatically constructing such pipelines is computationally difficult because operator sequences are combinatorial and end-to-end evaluation is expensive. Existing state-of-the-art (SOTA) Multi-DQN methods still face three key limitations: decoupled value estimators weaken long-horizon credit assignment, dataset context is only weakly injected into the policy, and exploration is inefficient in a sparse search space with many invalid states. To address these issues, we propose FlowPipe, a unified framework that formulates pipeline synthesis as conditional probabilistic flow generation over a directed acyclic graph. FlowPipe uses Conditional Generative Flow Networks (C-GFlowNets) with a Trajectory Balance objective to connect terminal validation","title":"FlowPipe: LLM-Enhanced Conditional Generative Flow Networks for Data Preparation Pipeline Construction","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-25T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.24679"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4205f85a9dbfe455280ef5b0a1655eed38d3b3c96095e861c4b0614ffc0922ad1192cc9ce618c518831532edb484349ed65d375a69d20530c6c8c634d6dfca0c","signer":"crovia.substrate","subject":{"observed_at":"2026-06-25T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.24679"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"bcb8adf3874b5a03483551c6f03999750f8a4aa3cb1fe7ec6838330c05978a07","leaf_index":247847,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"20890a8193d177672b05d4dba889afb0c0ef4dfadc4838c961f1fd1ae9d46b3b","side":"left"},{"sibling":"f30af1e9cf2db3a0ef42676ab785603a89b39f07d91bc1dbe83411cabebd1692","side":"left"},{"sibling":"0e92c2607b93c89aea33c45d83ae2407fb498942ca5ab170ee737999f743e306","side":"left"},{"sibling":"03ecebee352f245ab299f985b210c8c2c270d6840180865304f1c57a4eaaf87d","side":"right"},{"sibling":"9bcc1176e8d06ce6b48b92c7d8ac044a262c3ee8330592f466f39ee83f2b0db3","side":"right"},{"sibling":"d33d18ee6cc0c0a6a7604d918ab58d8868ed3e15a04c09d43e2f848c868f5e10","side":"left"},{"sibling":"04a5c087236178759d3641ae7a1f98c0237df27d3d245fa7f71cf83152e32896","side":"right"},{"sibling":"fd34dc262155fc820e683457e43a9a6ac4780d0d7acb55f8d250d6e34aac8216","side":"right"},{"sibling":"f1c209527e17fd1779db287808faab24759e4b119a0e00ac51b324128a47a586","side":"right"},{"sibling":"226eeefd4cac26b47f78a3ae7a84464ec61c8205c66c49d5a9780b83bb825fd9","side":"right"},{"sibling":"e3ba8de06c3c9f3ea70673a489a43265543dfe39b4f01afe43d581595abd7e34","side":"right"},{"sibling":"9a57f22647669db8e28640321fc83b219f5cc05f2a0fcee88bab0d42e2faddb0","side":"left"},{"sibling":"6219fab7afd19931e9188a852632b1af418f54d61e1a6fe0cb9d97b90c14bf46","side":"right"},{"sibling":"97afca3ebbdf205d326753e72068db4b0374a2039fe0e70ff3c5117bce4ab7c8","side":"right"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":247965,"merkle_root":"299142d480db2505acbf846802bce5a37d0de0218cf8f6a5483cf3ea85b97b86","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260625T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-25T05:37:55Z","sig_algorithm":"ed25519","signature":"05cc2e29ee1d5a741ff8f5b8ada11d63dfd94be9656249e36743f9add842801fc5051869e97eb7850ca7dd0c6d01cdd65410cc153fe65978b6f6da99ad19c60b","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_dc7075965e28a76b38ee430fdd71ff84a38c924c914f8ee52577a6b6e4c5aca0"}}