{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_93ccb39689842693af6665fce39c23252fbcaa9c76de3d46663df280b756707e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_93ccb39689842693af6665fce39c23252fbcaa9c76de3d46663df280b756707e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f967ec9983fc3ab955326e236b272d2d272c6c766fad924e6496bb67c2dcccb0","published":"Wed, 03 Jun 2026 00:00:00 -0400","receipt_hash":"f967ec9983fc3ab955326e236b272d2d272c6c766fad924e6496bb67c2dcccb0","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f967ec9983fc3ab955326e236b272d2d272c6c766fad924e6496bb67c2dcccb0","observed_at":"2026-06-03T04:43:57.136784Z","parent_run_hash":"62ae9c8eda846b00bc49666345b338d00756c5203438666c4c1fc694cc364b84","published":"Wed, 03 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.03660v1 Announce Type: new \nAbstract: Large language models are increasingly used as chemistry assistants, yet most chemistry benchmarks still score only final answers. This masks a critical failure mode: a model may output the correct molecule, product, or option while its reasoning violates chemical logic. Existing process-level evaluators are hard to scale because LLM judges and human step-level process annotation are costly, inconsistent, and vulnerable to hallucination. We introduce ChemCoTBench-V2, a rule-verifiable diagnostic benchmark for low-cost, auditable evaluation of structured, verifier-addressable chemical reasoning traces. It spans molecular understanding, molecule editing, molecular optimization, and reaction prediction, with 5,620 evaluation samples across 18 reporting tasks. Models must expose key intermediate steps in expert-designed templates, and those steps are checked with deterministic chemistry rules and, for closed-answer tasks, reference traces ra","title":"From Answers to States: Verifiable Process-Level Evaluation of Chemical Reasoning in Large Language Models","url":"https://arxiv.org/abs/2606.03660","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.03660v1 Announce Type: new \nAbstract: Large language models are increasingly used as chemistry assistants, yet most chemistry benchmarks still score only final answers. This masks a critical failure mode: a model may output the correct molecule, product, or option while its reasoning violates chemical logic. Existing process-level evaluators are hard to scale because LLM judges and human step-level process annotation are costly, inconsistent, and vulnerable to hallucination. We introduce ChemCoTBench-V2, a rule-verifiable diagnostic benchmark for low-cost, auditable evaluation of structured, verifier-addressable chemical reasoning traces. It spans molecular understanding, molecule editing, molecular optimization, and reaction prediction, with 5,620 evaluation samples across 18 reporting tasks. Models must expose key intermediate steps in expert-designed templates, and those steps are checked with deterministic chemistry rules and, for closed-answer tasks, reference traces ra","title":"From Answers to States: Verifiable Process-Level Evaluation of Chemical Reasoning in Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-03T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.03660"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:3715cfdd2e7e88cc315702729f3f71d29028cc74af8048c04fa9b3a29a5681b8949e2550599fe5dc7a5bf6073bd248611fb440fdbafaf4d57913ef2094509400","signer":"crovia.substrate","subject":{"observed_at":"2026-06-03T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.03660"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c003cb171bc4e7f2e8015846bf3389d91a68908496734a9a9aeb65e71f676c23","leaf_index":209068,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"ba87e3d8e0c31c11db0c9a193903511f2ddd08a29a15f2a4b2fa19bf07b73d55","side":"right"},{"sibling":"aeda6b649fa58cda86cb7a0de2b97f30e8bc9fd71156ef23bc0bd37c025f1716","side":"right"},{"sibling":"e5e14e53baa985b88a08e5212ddce2874923298f112b01f87618700d8c322cf8","side":"left"},{"sibling":"c7842d61be55dc5cc33f594a96af3621a7d6661b5ada79bad22f0ab51dc6225f","side":"left"},{"sibling":"8f344101640781283af134db303d3ac1b5aa655050a7e1bccf7dcb842073e2a6","side":"right"},{"sibling":"539a986e1a314056d614cc58138b39df2987d9391785174a6e8077bce59eb459","side":"left"},{"sibling":"774fc127f09c4116646d44b290dc856efd1c3b5371b99a54c99f0ed4cfbe2353","side":"right"},{"sibling":"eae7f3d160253f87d5f2b594663f65cef3187784faa6d32d0ec2a662c29c4d98","side":"left"},{"sibling":"9924ce3405008f0c3a3b79e9066c574929b10c2965e0cc083262d3ac8aae2cb3","side":"right"},{"sibling":"2abcec6d14b256f82b270a3141886de6129141dec525543607b760f02588a877","side":"right"},{"sibling":"2b6b45743f97ac502854e489ac38a3366f8ae7ede58a2728b45daadbf29e9d03","side":"right"},{"sibling":"a81babbd79ea0da9e030dd7f43bffb6519d214317decd50727bea4e78189d970","side":"right"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"8d3baa674a45fd8bc6d3e8d25298f4bec86c72fa8259d576c330d12955c7b4f7","side":"right"},{"sibling":"e32819d1eff909db08066d1703f2db3f091cddac19378b2c0c625ab11e3fdbc0","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":209569,"merkle_root":"c852efc8ad7dfffc196c71380f79e6398bcaf566974cbae7c9950b0f600d5bc8","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260603T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-03T05:37:57Z","sig_algorithm":"ed25519","signature":"1ca417607effc813341721cfdade0a2d2d4ba96a90d361dbc3e864b6991b90abc326ec41d9ba3e7110cc04adb29d7b8d95abea7ff2ab8c591b2f864ed7270c03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_93ccb39689842693af6665fce39c23252fbcaa9c76de3d46663df280b756707e"}}