{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d0dcfc5e3dc11bd2b352a8e5eb9eb41c47014a549a9afab0f83a9292b9916ea9","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d0dcfc5e3dc11bd2b352a8e5eb9eb41c47014a549a9afab0f83a9292b9916ea9","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6c2eaec59a25d3f990e86a19e167836705a4f7654861fd7658bd4c9dfa4e92de","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"6c2eaec59a25d3f990e86a19e167836705a4f7654861fd7658bd4c9dfa4e92de","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6c2eaec59a25d3f990e86a19e167836705a4f7654861fd7658bd4c9dfa4e92de","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.20068v1 Announce Type: new \nAbstract: While reinforcement learning from verifiable rewards (RLVR) typically has relied on a single binary verification signal, symbolic proof assistants in formal reasoning offer rich, fine-grained structured feedback. This gap between structured processes and unstructured rewards highlights the importance of feedback that is both dense and sound. In this work, we demonstrate that the Lean proof assistant itself can serve as a symbolic process oracle, supplying both outcome-level and fine-grained tactic-level verified feedback during training. Proof attempts are parsed into tactic sequences, and Lean's elaboration marks both locally sound steps and the earliest failing step, yielding dense, verifier-grounded credit signals rooted in type theory. We incorporate these structured rewards into a GRPO-style reinforcement learning objective with first-error propagation and first-token credit methods that balances outcome- and process-level advantage","title":"Process-Verified Reinforcement Learning for Theorem Proving via Lean","url":"https://arxiv.org/abs/2606.20068","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.20068v1 Announce Type: new \nAbstract: While reinforcement learning from verifiable rewards (RLVR) typically has relied on a single binary verification signal, symbolic proof assistants in formal reasoning offer rich, fine-grained structured feedback. This gap between structured processes and unstructured rewards highlights the importance of feedback that is both dense and sound. In this work, we demonstrate that the Lean proof assistant itself can serve as a symbolic process oracle, supplying both outcome-level and fine-grained tactic-level verified feedback during training. Proof attempts are parsed into tactic sequences, and Lean's elaboration marks both locally sound steps and the earliest failing step, yielding dense, verifier-grounded credit signals rooted in type theory. We incorporate these structured rewards into a GRPO-style reinforcement learning objective with first-error propagation and first-token credit methods that balances outcome- and process-level advantage","title":"Process-Verified Reinforcement Learning for Theorem Proving via Lean","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.20068"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:55be32551c4e62d84f55ca25f16b774a4f237c0a7834235aedf0f9b7102a2b55fa19ee54f50a03339a997326581bc478df90d9809158c5080ea15debdda31404","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.20068"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"4ead0188ac9857420eecc022ccb5ae953412a0e911efdc54f1d290f280fe3e48","leaf_index":235498,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"89809ac21154137c4f7d4a1698097aef432a703fc59e40b4b98d322ef156dda0","side":"right"},{"sibling":"cade756153571480594683daeec28eb6064492d9c8b9e0b16b7ab71377122f8e","side":"left"},{"sibling":"4d265e7f946dcf587a60ac4b93f7bfd6f88dce18df3e207413c06dc9c5a5e77f","side":"right"},{"sibling":"e31a2a541ab649ac2f13e838d6275c76a0c6ab1b9ad2f5b81f6a010077ec8f30","side":"left"},{"sibling":"bcd5aa5efe653b441d5f714c138f0ec81e2352e25c05bbf40f11618cd4c25dcb","side":"right"},{"sibling":"9d5b3a47759460fa8b5eceb45de221aafbf0cd60d3ee412f3351ff44394e520f","side":"left"},{"sibling":"696ac6bcf68966ee959b35539f0bed60b3216ccb53620defd29feb8dbcb2ac30","side":"left"},{"sibling":"95ba45f2fea2d822bc863962b7e478ea50086827d90fab88c0da065d454d7ac5","side":"left"},{"sibling":"00f27f149ddd2eb0d2cf60eaed9e1d55c662cf1cf69c95fbea03bba9d35f90c7","side":"left"},{"sibling":"797e0feb6bf826a55956c876711cd24824818c5a84a591f8cc06b95577b3405d","side":"left"},{"sibling":"f049d6e86b410f6f63921a6e3aa684efffa398fd23eed28245c43b156d404c5f","side":"left"},{"sibling":"9ec7f4f84de7e2057a02ac55686567b21acfd5beaecc7a84e9e48f3db17296c2","side":"right"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d0dcfc5e3dc11bd2b352a8e5eb9eb41c47014a549a9afab0f83a9292b9916ea9"}}