{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_be26bd3c3f8cfa67ea1a3e94710196c21c4bbbd0cd9a0378b9252cba81422272","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_be26bd3c3f8cfa67ea1a3e94710196c21c4bbbd0cd9a0378b9252cba81422272","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8cfe1d3d14e801d245bf677b565de72b6be712eeaaebacd0b615b05b5dc47152","published":"Tue, 26 May 2026 00:00:00 -0400","receipt_hash":"8cfe1d3d14e801d245bf677b565de72b6be712eeaaebacd0b615b05b5dc47152","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8cfe1d3d14e801d245bf677b565de72b6be712eeaaebacd0b615b05b5dc47152","observed_at":"2026-05-26T04:43:39.018238Z","parent_run_hash":"dca8dedd754ad6a1772113d6b97ee4f4ab9a0afbdeb44aaace5ff2d2446b164b","published":"Tue, 26 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2512.12576v3 Announce Type: replace-cross \nAbstract: While reinforcement learning has achieved impressive progress in language model reasoning, it is constrained by the requirement for verifiable rewards. Recent verifier-free RL methods address this limitation by utilizing the probabilities that LLMs generate reference answers as reward signals. However, these approaches typically sample reasoning traces conditioned only on the question. This design decouples reasoning-trace sampling from answer information, leading to inefficient exploration and incoherence between traces and final answers. In this paper, we propose \\textit{\\b{Co}upled \\b{V}ariational \\b{R}einforcement \\b{L}earning} (CoVRL), which bridges variational inference and reinforcement learning by coupling prior and posterior distributions through a hybrid sampling strategy. By constructing and optimizing a composite distribution that integrates these two distributions, CoVRL enables efficient exploration while preservi","title":"Coupled Variational Reinforcement Learning for Language Model General Reasoning","url":"https://arxiv.org/abs/2512.12576","vendor":"arxiv_cs_ai"},"summary":"arXiv:2512.12576v3 Announce Type: replace-cross \nAbstract: While reinforcement learning has achieved impressive progress in language model reasoning, it is constrained by the requirement for verifiable rewards. Recent verifier-free RL methods address this limitation by utilizing the probabilities that LLMs generate reference answers as reward signals. However, these approaches typically sample reasoning traces conditioned only on the question. This design decouples reasoning-trace sampling from answer information, leading to inefficient exploration and incoherence between traces and final answers. In this paper, we propose \\textit{\\b{Co}upled \\b{V}ariational \\b{R}einforcement \\b{L}earning} (CoVRL), which bridges variational inference and reinforcement learning by coupling prior and posterior distributions through a hybrid sampling strategy. By constructing and optimizing a composite distribution that integrates these two distributions, CoVRL enables efficient exploration while preservi","title":"Coupled Variational Reinforcement Learning for Language Model General Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-26T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2512.12576"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:965acf12e186629796b76e463925071dd2d2fe14bdd46b7677aa4037d5a366c7bcb89b74d58cee34ed28eb2b4babcef41dbdb98d486718f6f0bd70856f80b407","signer":"crovia.substrate","subject":{"observed_at":"2026-05-26T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2512.12576"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"b7e19649ec287061ed7dd8aa3db8f0d41fd66e6f90e1fede9b66e44bf5e08707","leaf_index":151912,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"dfd20d78668bbf4932b65f3a5249a6420244eed848fbb6d4008827f808eabe84","side":"right"},{"sibling":"6971c409dadb053925f64ac7dc676118cdb4d10d7e26b8a4b6651c24299f868a","side":"right"},{"sibling":"acc4e48b0dab68a61183fa42b7f13087d26ad450b1d8d7e3f9352d72d6a51534","side":"right"},{"sibling":"a1ca4863cf9d7f9bffe1c8e16a9b4cd60d5342a33a8542d085e6c955ed6523f6","side":"left"},{"sibling":"87915b70b83e592ab145305e3c9b94b78fe5c702c327429cd111c25abdad49fd","side":"right"},{"sibling":"e4c9c204f69afd56f712ff85cfef8caad2c7145db65e00ceb7f0d6a638ec6463","side":"left"},{"sibling":"7bd757323466f2d5e9a78f8308b6ddad9ce121c71e69a2be470e169f21e5583b","side":"left"},{"sibling":"b8013a4d5795f8230d3abb917aff567dadf5c7a3494daf94843cddd93f67ecf8","side":"right"},{"sibling":"6c8b632bc887ad8683d2d08a66a66babfd1e8e43de4b1b760ed3f3d35d5b02e1","side":"left"},{"sibling":"879666fab72e779ab55d0564eaabd64b00534fd6bba7f18412c7f31f61ffd09f","side":"right"},{"sibling":"f40ccedd90c323817e961adc0a2e2db82b8aabe192b6c9d5a373ff988987b207","side":"right"},{"sibling":"b85ea61ae405eed84392a7b6b1eee5536f5a38d6b04070638e23ec7b71e3443a","side":"right"},{"sibling":"d415e6939aee710631f5062799379b547d2c3e3d9a68f263bbb5a693285ab2ca","side":"left"},{"sibling":"e5893793e3591ed7f5e58ca94ffcfba46bb30f69fb1c25d5ba8ef49eb99f9126","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"e3eecaf996dbe7229a7bb1d234c97aea97a252c5f7c89f8547b6d091db0f0e40","side":"right"},{"sibling":"55bcbd4da3e20d93931f7e58673f10232e81a5b1514d7396cb4b71e8f95788d0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":152106,"merkle_root":"ac5182c6f3dd09931f2a689df5f4be36df7b55e55bcf106e195671f5ed55fd8f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260526T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-26T05:37:34Z","sig_algorithm":"ed25519","signature":"a401243fbd2c077d29a623c6ef616c78fbd5c8ce1930afa7af7165b2386e5a8cf15d5094983a1c962e71b27b911a3f3ce09c8cfab9393be2ce5ce8ee6513da06","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_be26bd3c3f8cfa67ea1a3e94710196c21c4bbbd0cd9a0378b9252cba81422272"}}