{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d923ddec48e83fcba7c9f656bb7ed25dc41458b0220c80c1d04be0e399bd6ea5","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d923ddec48e83fcba7c9f656bb7ed25dc41458b0220c80c1d04be0e399bd6ea5","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f8a77c25d3a922314b86e6019a6608ba18355a178e95a05c45d282b56c831984","published":"Fri, 10 Jul 2026 00:00:00 -0400","receipt_hash":"f8a77c25d3a922314b86e6019a6608ba18355a178e95a05c45d282b56c831984","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f8a77c25d3a922314b86e6019a6608ba18355a178e95a05c45d282b56c831984","observed_at":"2026-07-10T04:43:53.465232Z","parent_run_hash":"06997be187ba20932a2030c56de194579eacf484085252a25bee544eab183e91","published":"Fri, 10 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.08093v1 Announce Type: new \nAbstract: Large language models (LLMs) increasingly act as integrated data-science agents, combining abstract reasoning with advanced tool use. Yet the relevant benchmark landscape largely divides into symbolic causal reasoning benchmarks without realistic data analysis or data analysis benchmarks without a principled causal data-generating structure. Furthermore, existing causal evaluation datasets are often restricted to curated examples from existing sources, with diversity coming from limited templatized variations rather than from systematic generation of novel synthetic causal structures. We introduce CausalDS, a benchmark for evaluating causal reasoning in agentic data-science workflows. Each benchmark instance is a scene consisting of a sampled structural causal model (SCM) with generated observational data and an accompanying synthetic natural-language story grounded in a realistic domain. We optionally ground the composition of the bench","title":"CausalDS: Benchmarking Causal Reasoning in Data-Science Agents","url":"https://arxiv.org/abs/2607.08093","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.08093v1 Announce Type: new \nAbstract: Large language models (LLMs) increasingly act as integrated data-science agents, combining abstract reasoning with advanced tool use. Yet the relevant benchmark landscape largely divides into symbolic causal reasoning benchmarks without realistic data analysis or data analysis benchmarks without a principled causal data-generating structure. Furthermore, existing causal evaluation datasets are often restricted to curated examples from existing sources, with diversity coming from limited templatized variations rather than from systematic generation of novel synthetic causal structures. We introduce CausalDS, a benchmark for evaluating causal reasoning in agentic data-science workflows. Each benchmark instance is a scene consisting of a sampled structural causal model (SCM) with generated observational data and an accompanying synthetic natural-language story grounded in a realistic domain. We optionally ground the composition of the bench","title":"CausalDS: Benchmarking Causal Reasoning in Data-Science Agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-10T04:43:53Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.08093"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:28c0138cf5fe46346704e5076888612c18a26b8bbc442345438171952223e230371f65bfe9d0b8762eff9f8416dc9d8e009cf6191957f03461b847a1ea49da09","signer":"crovia.substrate","subject":{"observed_at":"2026-07-10T04:43:53Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.08093"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"56c9990a219c2fa3aeab502b06bd51c5e443577056b99d73caf308d3125a6e0f","leaf_index":299373,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"dc5726b25d10652cbad40596e9f6bf892a302a6e7978aebce164d839228b121d","side":"left"},{"sibling":"fb5836993b29c793a922121acf7eca4aeb22f807cfecb91297e039238dfe9a3c","side":"right"},{"sibling":"d00cd8874b1684736c849add91c69469fee6f456d2eb03ef9aa44d68e0e9c1a3","side":"left"},{"sibling":"64ae09d4d0ce41c1f3c2e90d702daf3ed684d653047cfaf7514ce3938c7d572d","side":"left"},{"sibling":"b912f62228207c2de122fede8032a078c5d3ebefb71f9c704cd06878702ee1d6","side":"right"},{"sibling":"f8d66ecc290a865e1dc8caf3d9a9176b9cc0743edec77de9e8ca456b72551a4c","side":"left"},{"sibling":"d7d1a57010639fd7da98ae2622b06b3bf73287b1959dbdbc7fa2102045b05bc1","side":"left"},{"sibling":"442ba9ebe9175420bc5e043f85ccde9011456ea96aa073c7096ba1d5b28a3706","side":"right"},{"sibling":"2c5ce75ad134aa442d7fbb5dc9968b0f2e48097c2ffaa2111d405de1c4a9c48b","side":"left"},{"sibling":"36a2c507282befad57202dd10278a66d37b402692f195a22d2275b3d1b2488d8","side":"right"},{"sibling":"f80e8d47e0860527b906fc2dba9a52609f7f623f2772479ae922cc019bab36d9","side":"right"},{"sibling":"cae83500ab2c25555aa6b5eaf9232696d15a868d91b34f7531dd955daadf70f7","side":"right"},{"sibling":"64dab64d51bdcb909e2a5e37efb8909d6704ecf824be484b5d2b60ee6e518890","side":"left"},{"sibling":"576f134a23c19a758ae5efd53016092a74b9900e878cf6eb4f3dab6be682b395","side":"right"},{"sibling":"3c65f53d7c3e4feba7c745e8df1327760ffa768eec84336db14d515a31731532","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"8728642cdb98496d916cc653f0d919d7eb2e89d0c529927c9e89091074ad584c","side":"right"},{"sibling":"8025674cb002a22ae243ca0c295c18c1d0ee119189ea88e08ac14a3a1468b8e3","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":299721,"merkle_root":"f7115d63193d3285ca28cb9f741ecea2513f9b3e492e785f97076f3cf8f9bb98","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260710T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-10T05:38:19Z","sig_algorithm":"ed25519","signature":"4eeedeb744885bff6523d66b1cb86bde62b36917696367addfd980d90b31011daece2e01b20608e4c0870ec31dd0bcb57d297447f01f153bf0716ad527142b00","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d923ddec48e83fcba7c9f656bb7ed25dc41458b0220c80c1d04be0e399bd6ea5"}}