{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_66afa3d9351c4a5cf4c584e46587ee5e2567b7d5c7ff18da6544e11765b4e81b","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_66afa3d9351c4a5cf4c584e46587ee5e2567b7d5c7ff18da6544e11765b4e81b","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"04199d4122fe9b8a71013217d2c6e1d3c984cb430e35bd3bf18d8695a56ea9b0","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"04199d4122fe9b8a71013217d2c6e1d3c984cb430e35bd3bf18d8695a56ea9b0","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"04199d4122fe9b8a71013217d2c6e1d3c984cb430e35bd3bf18d8695a56ea9b0","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.20227v1 Announce Type: new \nAbstract: Large Language Models (LLMs) have made significant progress in reasoning, particularly in deductive reasoning, which is crucial for high-stakes decision-making. As models improve, evaluation benchmarks should evolve to keep pace. However, existing benchmarks lack fine-grained control over logical complexity and struggle to balance semantic diversity with logical consistency.\n  To address these issues, we propose QMFOL, an automated framework for generating monadic first-order logic reasoning tasks with quantifiable and controllable complexity. It constructs formal logical structures using conjunction and disjunction patterns, enabling precise control over reasoning depth, width, label types, and distractors. These structures are then translated into natural language via LLMs, with logical consistency ensured through round-trip verification using an external prover. Based on our framework, we build QMFOLBench, a benchmark comprising 2880 ","title":"QMFOL: Benchmarking Large Language Model Reasoning via Quantifiable Monadic First-Order Logic Test Case Generation","url":"https://arxiv.org/abs/2606.20227","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.20227v1 Announce Type: new \nAbstract: Large Language Models (LLMs) have made significant progress in reasoning, particularly in deductive reasoning, which is crucial for high-stakes decision-making. As models improve, evaluation benchmarks should evolve to keep pace. However, existing benchmarks lack fine-grained control over logical complexity and struggle to balance semantic diversity with logical consistency.\n  To address these issues, we propose QMFOL, an automated framework for generating monadic first-order logic reasoning tasks with quantifiable and controllable complexity. It constructs formal logical structures using conjunction and disjunction patterns, enabling precise control over reasoning depth, width, label types, and distractors. These structures are then translated into natural language via LLMs, with logical consistency ensured through round-trip verification using an external prover. Based on our framework, we build QMFOLBench, a benchmark comprising 2880 ","title":"QMFOL: Benchmarking Large Language Model Reasoning via Quantifiable Monadic First-Order Logic Test Case Generation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.20227"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:c689561676f3b4fc7704cdb05edab8d1b4ce72bf75b2debd786c46613ddbea358436c223c09ddaf49a90e3664acac4b2a6d483a3b53a94af53342907708d5a05","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.20227"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"a90ad54f7dbf4a63d70b0d0c94c4abf4642bb5d43114fc828ced3c6e3c38894d","leaf_index":235510,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"872b3bbf039a9f68c3290c10cba1d1e4d084f564055443f5547a8fc8ec58059b","side":"right"},{"sibling":"c4f29a81a9ef2d67e2f8e006deb442b2b43d2c8f7d9ac4bd46659be60e93ea03","side":"left"},{"sibling":"23c1f71e692ba660b65c957d522f2a9db710f954e27d7f29a0c07e505de3cd49","side":"left"},{"sibling":"a4b88936b1283f387366fa1fd86a7b71aee15ebb9763cd37f00db6622e1f7fca","side":"right"},{"sibling":"a3ac6063a401de8c3ccad1f956c4abb3ff631101239ebb518dd4865bc94ef205","side":"left"},{"sibling":"9d5b3a47759460fa8b5eceb45de221aafbf0cd60d3ee412f3351ff44394e520f","side":"left"},{"sibling":"696ac6bcf68966ee959b35539f0bed60b3216ccb53620defd29feb8dbcb2ac30","side":"left"},{"sibling":"95ba45f2fea2d822bc863962b7e478ea50086827d90fab88c0da065d454d7ac5","side":"left"},{"sibling":"00f27f149ddd2eb0d2cf60eaed9e1d55c662cf1cf69c95fbea03bba9d35f90c7","side":"left"},{"sibling":"797e0feb6bf826a55956c876711cd24824818c5a84a591f8cc06b95577b3405d","side":"left"},{"sibling":"f049d6e86b410f6f63921a6e3aa684efffa398fd23eed28245c43b156d404c5f","side":"left"},{"sibling":"9ec7f4f84de7e2057a02ac55686567b21acfd5beaecc7a84e9e48f3db17296c2","side":"right"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_66afa3d9351c4a5cf4c584e46587ee5e2567b7d5c7ff18da6544e11765b4e81b"}}