{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_9a1c73a5bb0601df63cab32fbd3f9cc49edea5aafa9f14dab46edcafe1fc62d7","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_9a1c73a5bb0601df63cab32fbd3f9cc49edea5aafa9f14dab46edcafe1fc62d7","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0e8114832c1bc35fe5b908001c0592ffebaba50e7eef1ea613fa04613163f0fe","published":"Mon, 15 Jun 2026 00:00:00 -0400","receipt_hash":"0e8114832c1bc35fe5b908001c0592ffebaba50e7eef1ea613fa04613163f0fe","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0e8114832c1bc35fe5b908001c0592ffebaba50e7eef1ea613fa04613163f0fe","observed_at":"2026-06-15T04:43:09.998079Z","parent_run_hash":"ded7a5fa7968821af82d6d8d24b2c1f7e7d776433180609016edbdee95e78c1a","published":"Mon, 15 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.13782v1 Announce Type: new \nAbstract: Large Language Models (LLMs) have made notable progress in automated theorem proving, yet existing formal benchmarks remain limited in both mathematical coverage and difficulty. Most are concentrated in areas that are easier to formalize, such as algebra and elementary number theory, and provide limited coverage of subfields that require deeper reasoning, including mathematical analysis. To address this gap, we introduce MA-ProofBench, to the best of our knowledge, the first formal theorem-proving benchmark dedicated to Mathematical Analysis. The benchmark contains 200 formalized theorems covering 6 core topics and 27 subcategories, including measure and integration theory, complex analysis, and functional analysis. The problems are divided into two difficulty levels, an undergraduate level (Level I, 100 problems) and a Ph.D. qualifying level (Level II, 100 problems), to evaluate how well LLMs perform formal reasoning at different mathem","title":"MA-ProofBench: A Two-Tiered Evaluation of LLMs for Theorem Proving in Mathematical Analysis","url":"https://arxiv.org/abs/2606.13782","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.13782v1 Announce Type: new \nAbstract: Large Language Models (LLMs) have made notable progress in automated theorem proving, yet existing formal benchmarks remain limited in both mathematical coverage and difficulty. Most are concentrated in areas that are easier to formalize, such as algebra and elementary number theory, and provide limited coverage of subfields that require deeper reasoning, including mathematical analysis. To address this gap, we introduce MA-ProofBench, to the best of our knowledge, the first formal theorem-proving benchmark dedicated to Mathematical Analysis. The benchmark contains 200 formalized theorems covering 6 core topics and 27 subcategories, including measure and integration theory, complex analysis, and functional analysis. The problems are divided into two difficulty levels, an undergraduate level (Level I, 100 problems) and a Ph.D. qualifying level (Level II, 100 problems), to evaluate how well LLMs perform formal reasoning at different mathem","title":"MA-ProofBench: A Two-Tiered Evaluation of LLMs for Theorem Proving in Mathematical Analysis","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-15T04:43:09Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.13782"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:3fa86d3d04a4de3d32db20cd24ccc320d1376fe4f1855aa45ab244012ad3c2da89483acc700ff8f1194db4f49af94bf2902d2879608a562864c32a41e2b0a20a","signer":"crovia.substrate","subject":{"observed_at":"2026-06-15T04:43:09Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.13782"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"9724b835c3054b70bfcf3fe6e23f5e003081347b3540d3e2343f9e5c41dadf0b","leaf_index":229934,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"21b9f8e9138f9ac80eb731971934eaeef64842aa9677417117dacb18fabb176f","side":"right"},{"sibling":"52b366a3341faf24850aefe1e492c65441927e05ce046929416a60a08989736b","side":"left"},{"sibling":"03848a34ce29fca41e5ef1a1d9e380e1e68a6c0ac9acfec61070f3a4b1389c77","side":"left"},{"sibling":"bcda13e23e00c78c972972aa8572f684af00d29f02e78c19add245dcb2c711ad","side":"left"},{"sibling":"16620991a0f27b14ffedb603c0eb162b262a73e0a05bd91fbefd99113f584539","side":"right"},{"sibling":"e6cdecdfc7ac8cd05b0d8e3c44fb53f93f9a4b64780bdd1cc70e657d50712a06","side":"left"},{"sibling":"6bde248411f8ec3173e97ec3895054895c25f208bf85a22fd09c8fdf4911671b","side":"right"},{"sibling":"bc1002bb7e3da047b8a7c8f3990db61fb56edf499868605dc0c31aa0dee387b7","side":"right"},{"sibling":"14c50c43949e1ad41f149ffea691627d3f715c5861766c693b9fbac9d03b0d90","side":"right"},{"sibling":"74897e850164dddc689c3c65b33f9bae0268ab0bf429867a4e193d9b9b685040","side":"left"},{"sibling":"bde25d7e94e64717e426a97f6fcb4907e92b5c61fc89d92d7e0947a2249c3f6b","side":"right"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_9a1c73a5bb0601df63cab32fbd3f9cc49edea5aafa9f14dab46edcafe1fc62d7"}}