{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_2b9413def3266b3d5ea6c27d2642107d82ac84e29c0c339e2d0e55b1640273fb","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_2b9413def3266b3d5ea6c27d2642107d82ac84e29c0c339e2d0e55b1640273fb","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"42a9723de95ae7dc50f3f4b1792e538d005c54f4e76807ed8e25dd6029741875","published":"Mon, 29 Jun 2026 00:00:00 -0400","receipt_hash":"42a9723de95ae7dc50f3f4b1792e538d005c54f4e76807ed8e25dd6029741875","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"42a9723de95ae7dc50f3f4b1792e538d005c54f4e76807ed8e25dd6029741875","observed_at":"2026-06-29T04:44:03.414425Z","parent_run_hash":"36b5ab5c57ae76dc9e1a863501c4d38f172868cfae31b4ba5baf3caffaafb2c4","published":"Mon, 29 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.20094v2 Announce Type: replace \nAbstract: As large language models (LLMs) witness increasing deployment in complex, high-stakes decision-making scenarios, it becomes imperative to ground their reasoning in causality rather than spurious correlations. However, strong performance on traditional reasoning benchmarks does not guarantee true causal reasoning ability of LLMs, as high accuracy may still arise from memorizing semantic patterns instead of analyzing the underlying true causal structures. To bridge this critical gap, we propose a new causal reasoning benchmark, CausalFlip, designed to encourage the development of new LLM paradigm or training algorithms that ground LLM reasoning in causality rather than semantic correlation. CausalFlip consists of causal judgment questions built over event triples that could form different confounder, chain, and collider relations. Based on this, for each event triple, we construct pairs of semantically similar questions that reuse the ","title":"CausalFlip: A Benchmark for LLM Causal Judgment Beyond Semantic Matching","url":"https://arxiv.org/abs/2602.20094","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.20094v2 Announce Type: replace \nAbstract: As large language models (LLMs) witness increasing deployment in complex, high-stakes decision-making scenarios, it becomes imperative to ground their reasoning in causality rather than spurious correlations. However, strong performance on traditional reasoning benchmarks does not guarantee true causal reasoning ability of LLMs, as high accuracy may still arise from memorizing semantic patterns instead of analyzing the underlying true causal structures. To bridge this critical gap, we propose a new causal reasoning benchmark, CausalFlip, designed to encourage the development of new LLM paradigm or training algorithms that ground LLM reasoning in causality rather than semantic correlation. CausalFlip consists of causal judgment questions built over event triples that could form different confounder, chain, and collider relations. Based on this, for each event triple, we construct pairs of semantically similar questions that reuse the ","title":"CausalFlip: A Benchmark for LLM Causal Judgment Beyond Semantic Matching","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-29T04:44:03Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.20094"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:8bd1561bdf634170e7a925a32da83fe9a7e8f5d372333b9e99f979e3439d3fcfd0376c223cb484a94496783b7bda05bb4c2b7c26004468b2914a69e8b3e81c09","signer":"crovia.substrate","subject":{"observed_at":"2026-06-29T04:44:03Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.20094"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d46b3d75d88ce79b8e754646125096aadda54eb4dd7e39538c9747c8beac8873","leaf_index":261457,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fea4c4ca53bc29b56aa7dae67b24e99a08b86669e142c03be58eb24bceb5a7f3","side":"left"},{"sibling":"d2f9fdd85769cb5c2e32cf938e029f9d72f4ac636829107f64a08936020e9767","side":"right"},{"sibling":"138728ca0084a4cbf7a4833027c0d2709c78329cad0dac03488da76cb52911d6","side":"right"},{"sibling":"d19f3dff6933318c36d1ba833dfb8e79ce08a1a3326428cd667d2180804aaf88","side":"right"},{"sibling":"7e04f95e9085320e53f546d0a4ec1e70345e44bfd0153b0509729e7bcc572563","side":"left"},{"sibling":"6964c7fbbb2e4940d0c2ddfcfef79a2a9e7d53aa2f229440f53a7577177c532e","side":"right"},{"sibling":"a600c67ab28d0232fe4727e4641fc665bde749779b50edf849b1a9c8287bc809","side":"left"},{"sibling":"4b4bda5fa1fb23989b8f6f192c36dadc8266c378c69d42945af35c9d6ab81a10","side":"right"},{"sibling":"6d1df65d14253ed3a13c03e25aa9adcd9a091d8062a82e8586bcaee74013b90e","side":"left"},{"sibling":"9f9daa9d12e65b219f34c92aec45450536b79a42b8892050d66961432ae28ed1","side":"right"},{"sibling":"b5725d7b0807dc6da32d9788f20057fa8726be38d30a9ebdabc605ae92739122","side":"left"},{"sibling":"e321b2cac14cbe28f76ccb7938249a40ff60cd5d2128b5634be046ea10e984b8","side":"left"},{"sibling":"5900dc6c7d13855af9d0385baf1691ec386df33e450c422af1cabe0a36e40ad8","side":"left"},{"sibling":"ae636ddee98c71ab7a7dc55ddfab70c7f710a2b6abfdf7a8b5d16a4017d1c0d1","side":"left"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":261662,"merkle_root":"aa8865c239aa2eb6c8aa7c6250f56b3cd5709854a8a07f6a29eb4ddd8802cb6f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260629T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-29T05:38:02Z","sig_algorithm":"ed25519","signature":"476329233e82fb35fba2552ddc5d1d75b2bdd8513bbd281e9c40a0b8e475df374a62dcd8b456b0c5e8815984f5b4bf0983ae95d2cf4d7412ebb13a433b933c0a","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_2b9413def3266b3d5ea6c27d2642107d82ac84e29c0c339e2d0e55b1640273fb"}}