{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_7763c1530de12eb60d9452a7badb47b45ecf990e869ea0cbce6a888e8cc9160f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_7763c1530de12eb60d9452a7badb47b45ecf990e869ea0cbce6a888e8cc9160f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"875cc826a684f789b8d25ad1e63607aafc7dcd987440a1d185e575f70550dabd","published":"Wed, 10 Jun 2026 00:00:00 -0400","receipt_hash":"875cc826a684f789b8d25ad1e63607aafc7dcd987440a1d185e575f70550dabd","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"875cc826a684f789b8d25ad1e63607aafc7dcd987440a1d185e575f70550dabd","observed_at":"2026-06-10T04:43:37.461885Z","parent_run_hash":"23aff1a6f676ba7ca33f70f4ddfae1dd282fb86104d577ce9be510d81a94c5dc","published":"Wed, 10 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.10799v1 Announce Type: new \nAbstract: Large Language Models (LLMs) struggle to rigorously verify complex mathematical proofs. Standard global evaluation approaches suffer from \"context poisoning,\" in which superficially plausible statements mask subtle logical flaws, leading to hallucination or over-skepticism. To address this, we shift from global evaluation to strict step-level verification: our framework maintains detailed context for each deduction step and strictly constrains the sources of applied theorems. We evaluate on a carefully curated adversarial diagnostic suite of research-level proofs drawn from the FirstProof challenge. A systematic ablation study demonstrates that these deductive constraints are indispensable, as unconstrained global prompting consistently fails to localize subtle logical errors. Beyond outperforming global evaluation, our approach fundamentally alters the failure taxonomy. Error analysis reveals that, rather than exhibiting severe logical ","title":"Evaluating Research-Level Math Proofs via Strict Step-Level Verification","url":"https://arxiv.org/abs/2606.10799","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.10799v1 Announce Type: new \nAbstract: Large Language Models (LLMs) struggle to rigorously verify complex mathematical proofs. Standard global evaluation approaches suffer from \"context poisoning,\" in which superficially plausible statements mask subtle logical flaws, leading to hallucination or over-skepticism. To address this, we shift from global evaluation to strict step-level verification: our framework maintains detailed context for each deduction step and strictly constrains the sources of applied theorems. We evaluate on a carefully curated adversarial diagnostic suite of research-level proofs drawn from the FirstProof challenge. A systematic ablation study demonstrates that these deductive constraints are indispensable, as unconstrained global prompting consistently fails to localize subtle logical errors. Beyond outperforming global evaluation, our approach fundamentally alters the failure taxonomy. Error analysis reveals that, rather than exhibiting severe logical ","title":"Evaluating Research-Level Math Proofs via Strict Step-Level Verification","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-10T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.10799"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b9e106d1a26ac45711d6e819ddbcfba95d0c213469ebc9be3e58448f54fc3620a63a7349a3d464bb68eca0ae2b8c5d7c65f1f1431cb232e259dbc3e830491d09","signer":"crovia.substrate","subject":{"observed_at":"2026-06-10T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.10799"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"0b87a3cc8dd9f1605f5098ed2102219755203af9d24f4424bfd1069d789fcc45","leaf_index":225875,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"cdaa82ac63fc26e0b1625fa8ef14b3188e9371631a7fe140d7b3a12ac143fc15","side":"left"},{"sibling":"beb102ab11bb3c1104187120017bdea25629cafb1701589133069320f366bae3","side":"left"},{"sibling":"4c6fb64bf78466b1ed47d13dfe350452264b384183f046c2f132b1a17e8a7d0e","side":"right"},{"sibling":"b2f82ef98f6610613510df6750c7a1186771e8e8b3ba22b4a7ac02dd14cf5122","side":"right"},{"sibling":"8e7db903c2bbfc27543c0f79dd7578deaaa76e8801e776250993cd141f5e0305","side":"left"},{"sibling":"2e627a04aead5091204cc45a543ab572843c0df27707aa2624afc983f9387e00","side":"right"},{"sibling":"b2b764628ccd5a7c6a8404965429a36513804c4f359049b6a0f70892a8a53490","side":"left"},{"sibling":"b42fc5e6a220ca97d9fe75065fb0dafd72b915ddda12e63083678d3813da9eee","side":"right"},{"sibling":"ac698c3a6027f35b513fe892f166e70344e855625238cbb581d0bf06c131db80","side":"right"},{"sibling":"a4d17aefe58175050dc159af6246658fcf1c9f3ed57aacf1b350fc3261de4e69","side":"left"},{"sibling":"280b980aa0c7756b0b0cb22658f26466d36f0e70fbc3312cd2311d9898e30b8f","side":"right"},{"sibling":"c98954d4b658b1dda60fe52576fcf9bf21a2d49c67fb63f8c30f16ab5f721938","side":"right"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_7763c1530de12eb60d9452a7badb47b45ecf990e869ea0cbce6a888e8cc9160f"}}