{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_0665bb4a10d4350e4fb379f9ff00e56ab3176e90346c329353bc4b43d0e1ca91","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_0665bb4a10d4350e4fb379f9ff00e56ab3176e90346c329353bc4b43d0e1ca91","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"def1e4216220e95d9c74754be1ec9ea6f1d53e73830db4d05ecae06bfde6a5a0","published":"Wed, 01 Jul 2026 00:00:00 -0400","receipt_hash":"def1e4216220e95d9c74754be1ec9ea6f1d53e73830db4d05ecae06bfde6a5a0","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"def1e4216220e95d9c74754be1ec9ea6f1d53e73830db4d05ecae06bfde6a5a0","observed_at":"2026-07-01T04:43:38.812993Z","parent_run_hash":"0e10ec7d671a375a4e18d8645653df2b4352c82fe16fae2a400af6f7634d6699","published":"Wed, 01 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.30851v1 Announce Type: cross \nAbstract: Improving the reliability of large language models (LLMs) at inference time is a central challenge in structured reasoning tasks such as Text-to-SQL. Common test-time inference strategies, including Best-of-N sampling and Majority Voting, rely on heuristic signals such as execution success or output frequency, which provide limited semantic discrimination across candidate outputs. In this work, we study Outcome Reward Models (ORMs) as learned semantic scoring functions for test-time verification in Text-to-SQL. While ORMs have been previously explored for test-time scaling and alignment, their application to structured query generation remains underexplored. We introduce GradeSQL, a scalable framework for training task-specific ORMs via automated candidate generation and execution-based labeling, enabling verifier training without manual annotation. We integrate ORMs into a verification-driven Best-of-N pipeline and evaluate our approa","title":"Test-Time Verification for Text-to-SQL via Outcome Reward Models","url":"https://arxiv.org/abs/2606.30851","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.30851v1 Announce Type: cross \nAbstract: Improving the reliability of large language models (LLMs) at inference time is a central challenge in structured reasoning tasks such as Text-to-SQL. Common test-time inference strategies, including Best-of-N sampling and Majority Voting, rely on heuristic signals such as execution success or output frequency, which provide limited semantic discrimination across candidate outputs. In this work, we study Outcome Reward Models (ORMs) as learned semantic scoring functions for test-time verification in Text-to-SQL. While ORMs have been previously explored for test-time scaling and alignment, their application to structured query generation remains underexplored. We introduce GradeSQL, a scalable framework for training task-specific ORMs via automated candidate generation and execution-based labeling, enabling verifier training without manual annotation. We integrate ORMs into a verification-driven Best-of-N pipeline and evaluate our approa","title":"Test-Time Verification for Text-to-SQL via Outcome Reward Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-01T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.30851"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:10f63f2f65310a9a1418346b10694ff7b71134c12da2a9ec3a162ba6c20ace251f57b847dcfb5af74357f917b941593ae04184f6327b9ac142a459510393b00b","signer":"crovia.substrate","subject":{"observed_at":"2026-07-01T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.30851"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"801dc5140074e55975537f33e6de685b570e5e971079f79d722c5af2d1018d73","leaf_index":268510,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"2765c24f045fc82157427c262344ff0dceac21fa848326787872173afc2e1a52","side":"right"},{"sibling":"8cb5e01276705c091433ece6173cbe4709dad71193be0bd9d043382a49c10917","side":"left"},{"sibling":"85922de78c545f766dca3551e5f6e249c1d9c06ef9c2a3ee3fbd01d72e098520","side":"left"},{"sibling":"4d396d13b0940ae9e6db5d4d26e59ef0d828a86bdda5a8811d40f29e6f895aec","side":"left"},{"sibling":"8237f11ae5aed9d85a2f5614e53b6e6f20a08d7623c319d0fe6e042bd137d849","side":"left"},{"sibling":"1f57a328336cba4f64ebda75a50e218bfbd4e765f97b60048ff838beb1cadf1d","side":"right"},{"sibling":"8325907723322726e28223e7a58209725d2c897da06068f32115d78744fe46e1","side":"left"},{"sibling":"373441c77391345a6bdd59f6d24909be6fec81e993e9a7050109db1522722b95","side":"left"},{"sibling":"25b5e7b84bfdaf7764d4179af698e218f4b2858d756d4fd6c30f7dc8f562c1b9","side":"right"},{"sibling":"bbd9a20451913e8c7b910f616d5661d9b281a94af70d201ba50b3112d429521b","side":"right"},{"sibling":"85af80e45748c1e0da1e2f42d9d66da8996016888a034f20eb17ff0a73b69eab","side":"right"},{"sibling":"4575fde969d1d9a2984cc01a37ac8441238f74527d42874272dc5582dadebb4f","side":"left"},{"sibling":"f536de281672cbf0b583a3dc46faef1de2823b72bd9265c7b60e889131cc268d","side":"left"},{"sibling":"95b8b0f67237052a17c41fa8cbdce2b79bcb5aeeba3fdb239d4c4497e598115d","side":"right"},{"sibling":"c2f351f771cee329448890504d9436792ba482e50250ca9a19289311131f96c8","side":"right"},{"sibling":"c39bfb2e911ca37ae997690bfc04128ae32e6806ee1cb3908781a7e6685022a0","side":"right"},{"sibling":"a101b4c60ef6854ac3d750eef02d8e2e06c153284b5ecb7111302f97eb129797","side":"right"},{"sibling":"eae2a3de5cb35455ad60125e196cfadaba8a590c53146e95028469f53f70349c","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":268860,"merkle_root":"d098f25810d0569730b6c0e170d57f329a70359d483b47874b5f2fc51d23de65","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260701T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-01T05:38:05Z","sig_algorithm":"ed25519","signature":"64f37f0e3df0556baa55b924a736cab8005643fa0ab65b503f22407de30eab293e6e0bd62904270fda38d05084bb350c8c0d0aadb2c5a6bae5f1c54598e6d30c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_0665bb4a10d4350e4fb379f9ff00e56ab3176e90346c329353bc4b43d0e1ca91"}}