{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_e3ab4cda99f9555518332243f4fe6f74fa301c80a5d1dfa253ebed5b63471697","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_e3ab4cda99f9555518332243f4fe6f74fa301c80a5d1dfa253ebed5b63471697","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6ac2c8cb3a2bb422ea993b9fe4434a98e0648f86e749f5411cfabe8d07f93dce","published":"Thu, 04 Jun 2026 00:00:00 -0400","receipt_hash":"6ac2c8cb3a2bb422ea993b9fe4434a98e0648f86e749f5411cfabe8d07f93dce","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6ac2c8cb3a2bb422ea993b9fe4434a98e0648f86e749f5411cfabe8d07f93dce","observed_at":"2026-06-04T04:43:08.243501Z","parent_run_hash":"298818240313a3c9ce3ace3750dbf55845013dc3bc59aeced6331eccefdb61ac","published":"Thu, 04 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.04503v1 Announce Type: cross \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has greatly advanced large reasoning models (LRMs), but it requires timely training on a huge fully-annotated dataset. To this end, data-efficient RLVR methods have been widely studied from two perspectives: (i) data selection methods identify a small subset of \"golden\" samples that yield near-full-data performance, but they rely on a pre-existing pool of labeled data. (ii) unsupervised RLVR methods train the model using its own internal supervision signals on large-scale unlabeled data, yet they exhibit suboptimal performance. Accordingly, we investigate the \"pick in the dark\" setup for RLVR, which aims to select, without prior supervision, unlabeled samples that are most beneficial for training and worthy of annotation. Through systematic analysis, we demonstrate that smart picks hinge on a well-calibrated uncertainty estimator to enable strategic partitioning of data for adaptive","title":"Smart Picks in the Dark: Towards Efficient RLVR for Reasoning via Tracing Metacognitive Pivots","url":"https://arxiv.org/abs/2606.04503","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.04503v1 Announce Type: cross \nAbstract: Reinforcement learning with verifiable rewards (RLVR) has greatly advanced large reasoning models (LRMs), but it requires timely training on a huge fully-annotated dataset. To this end, data-efficient RLVR methods have been widely studied from two perspectives: (i) data selection methods identify a small subset of \"golden\" samples that yield near-full-data performance, but they rely on a pre-existing pool of labeled data. (ii) unsupervised RLVR methods train the model using its own internal supervision signals on large-scale unlabeled data, yet they exhibit suboptimal performance. Accordingly, we investigate the \"pick in the dark\" setup for RLVR, which aims to select, without prior supervision, unlabeled samples that are most beneficial for training and worthy of annotation. Through systematic analysis, we demonstrate that smart picks hinge on a well-calibrated uncertainty estimator to enable strategic partitioning of data for adaptive","title":"Smart Picks in the Dark: Towards Efficient RLVR for Reasoning via Tracing Metacognitive Pivots","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-04T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.04503"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:e3d5bdbb7b53aa5781c297d68675f95da7aabc9ce5582d171b321bc6400b2666a4cf76c48559c92b1d0792b9ae1fb3b26352c907565d9a140ecf4d332dd92f02","signer":"crovia.substrate","subject":{"observed_at":"2026-06-04T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.04503"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"b3af055a6b8965438a8e3397de0358de7720dba1aab439898b00bca867119498","leaf_index":212694,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c1d209e36b341ae53efb0d188b4729fbad79c6ea40390633dd35d793759de956","side":"right"},{"sibling":"34c1420c076b1ad8037aff6bbb3e2c745b61effc37458b3b7e89bdc7ce3a3cae","side":"left"},{"sibling":"b234e6d9193efc5d4eac8c88697728fcac7aba0678f87045dd31d3345d991506","side":"left"},{"sibling":"5b36effdda2dfaadc110d932900f02fe0f586b88a94f9739ba8709f22de6fbf0","side":"right"},{"sibling":"d40f0e95c044ae3bf08a999c1eaef75e053b543f6127d2373852c876c6cd8ff5","side":"left"},{"sibling":"c5d50cd6db693cc08f4200be7f161b457ba2158d77c3b2935c21e80bddc41861","side":"right"},{"sibling":"aa2a087d744cdb0fc9ed84d70ba0527bef5f726c738f58a25cae0e735775683f","side":"left"},{"sibling":"2943aa232ba1099978a3aeee29e3e0c39faa6231597512d9c82160e7a5a2db32","side":"left"},{"sibling":"c78f0d662697b816291c956749c41d06c16e4dc4c3d390f9a90590e3f2821765","side":"right"},{"sibling":"cfb460164a914d1f96a36aa17124b45bb5417829d2adbe5ca48375dbd842ec44","side":"left"},{"sibling":"a84ebc8e894a9893ee34d1afc942d15f28243b926f1e5f9d7f10a3de1e795262","side":"left"},{"sibling":"f8f6bd9da448fa097e2115b71146f61691d9f8807aca291c87c633192a7224e9","side":"left"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"422bcf7e281ca3a607f3726a5e6b8fabdb85e8b32199b4a86356998e260a0b34","side":"right"},{"sibling":"54a99163a4a62374c3ca6fb46294222f1d4b1a9d0b636e27256b0e093e98239a","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":213053,"merkle_root":"19d104b92c4d7299881c447fb8422611fc9cb8615d37a538959e34a2da7ef55f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260604T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-04T05:37:49Z","sig_algorithm":"ed25519","signature":"630748e88645187aa3b4d4cb8c872cc146180d0301e4c6656f15f99081d7776432715cea2a997b32b7b3a5216f215d19c1b536c0b093100ec843fa0bd65f2101","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_e3ab4cda99f9555518332243f4fe6f74fa301c80a5d1dfa253ebed5b63471697"}}