{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_0317e8de5163a14cef0e9ee55c10748596dde21b70124a281b68c11211934fab","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_0317e8de5163a14cef0e9ee55c10748596dde21b70124a281b68c11211934fab","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"eecb0e83eefb0855b2e3c0d2ba0884a39e4933fbcae16fdc660c900e910e4867","published":"Fri, 24 Jul 2026 00:00:00 -0400","receipt_hash":"eecb0e83eefb0855b2e3c0d2ba0884a39e4933fbcae16fdc660c900e910e4867","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"eecb0e83eefb0855b2e3c0d2ba0884a39e4933fbcae16fdc660c900e910e4867","observed_at":"2026-07-24T04:43:08.456021Z","parent_run_hash":"b018378f86139a28e6209ec008b31c1282cd1b5c1632dbd43b054c17aa88ab96","published":"Fri, 24 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.25077v2 Announce Type: replace \nAbstract: Weak-to-strong alignment offers a promising route to scalable supervision, but it can fail when a strong model becomes confidently wrong on examples that lie in the weak model's blind spots. Understanding such failures requires going beyond aggregate accuracy, since weak-to-strong errors depend not only on whether the strong model disagrees with the weak model, but also on how confidence and uncertainty are distributed across examples. In this work, we analyze weak-to-strong alignment through a bias--variance--covariance lens that connects misfit theory to practical post-training pipelines. We derive a misfit-based upper bound on weak-to-strong population risk and study its empirical components using continuous confidence scores. We evaluate four weak-to-strong pipelines spanning supervised fine-tuning (SFT), reinforcement learning from human feedback (RLHF), and reinforcement learning from AI feedback (RLAIF) on the PKU-SafeRLHF and","title":"Evaluating Risks in Weak-to-Strong Alignment: A Bias-Variance Perspective","url":"https://arxiv.org/abs/2604.25077","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.25077v2 Announce Type: replace \nAbstract: Weak-to-strong alignment offers a promising route to scalable supervision, but it can fail when a strong model becomes confidently wrong on examples that lie in the weak model's blind spots. Understanding such failures requires going beyond aggregate accuracy, since weak-to-strong errors depend not only on whether the strong model disagrees with the weak model, but also on how confidence and uncertainty are distributed across examples. In this work, we analyze weak-to-strong alignment through a bias--variance--covariance lens that connects misfit theory to practical post-training pipelines. We derive a misfit-based upper bound on weak-to-strong population risk and study its empirical components using continuous confidence scores. We evaluate four weak-to-strong pipelines spanning supervised fine-tuning (SFT), reinforcement learning from human feedback (RLHF), and reinforcement learning from AI feedback (RLAIF) on the PKU-SafeRLHF and","title":"Evaluating Risks in Weak-to-Strong Alignment: A Bias-Variance Perspective","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-24T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.25077"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ffa710ae54fd0ee10a175fce62243c449d46b5756d53b26d36079f08bd8ba8b3f41a36ab07e193417bd06e111bec5fc6064ac32d1d794cbfc3eb53596128930e","signer":"crovia.substrate","subject":{"observed_at":"2026-07-24T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.25077"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"ff144c6e2b50eef911c9d12219c3796e9b1b639a808ab40011a92fa4e6d49cc1","leaf_index":347216,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7923c0c90f39b573266720eac197a8423114fee97f233594c31da351d3f7a316","side":"right"},{"sibling":"7a5a524b90f1f6220c78130d3b144339c7a92a3f513d3d9afda8fccb83b247a5","side":"right"},{"sibling":"bd700d474cf5926271668db237816791b40613ffab5218c44069f44386e72542","side":"right"},{"sibling":"23725355552aef9f43d18bee7fa4a6c15652973a472cfdad6315abd11d12e661","side":"right"},{"sibling":"dbb495c663955cbcf6e9993cd64f61444ca69fd2c1ae1ded3cd67f1a3ca90ef9","side":"left"},{"sibling":"df047800e24edab2e706d2da8a55af57b632bfb1e158709b284e530c20f38633","side":"right"},{"sibling":"22949f5b6c661121ddf62ce35bcf81cd2664e41e1c6aeed348c449bca292ce6b","side":"left"},{"sibling":"aa4b293ba10895bb7f8b28f0f04360b53a8fc5a7523d9a560dbb7b29d003643e","side":"right"},{"sibling":"759f431c97765cb7fb12d38ee64fd6a87076abf1b63c87d7c8c172d57b29084d","side":"right"},{"sibling":"b9cb83ecb59812b3b9270c365951fa79dead288c3923b6cf1f83ffb911b27405","side":"right"},{"sibling":"e6c9083cd0939b38f691f619de2054068654e9c77f7c7ab051e0d48b77339e5d","side":"left"},{"sibling":"d3139af8c5ce235438e1c69e4b7afa44ba09129fd86674968434f23e546f423e","side":"left"},{"sibling":"cb89775a838ee16d10fc8da3213420c2012b4d96e8d55cd49939b0887a4b92d3","side":"right"},{"sibling":"252d30ea8052c3bb6b40bc5cc29fc9b9725343d212f84c08fbbae4215a125b00","side":"right"},{"sibling":"f3e45bceed774d2402fa45d41ff5190f295823bd2f216eb90157884150034693","side":"left"},{"sibling":"3cfa2102c0224815c6f3bf73e6710e24103f43f7bf5da1ca2abad1416d9c0890","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"e871fd7edf9b2ad89bce1609a028f5225eea4d14372169bac242420830f86530","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":347413,"merkle_root":"9efe042c5dd6583dfd3b6a58fbfc289807f60bcf2bd2927f10488a54a8ba11fc","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260724T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-24T05:38:42Z","sig_algorithm":"ed25519","signature":"8633c55f558d42994850505218b2862c6134bad2b1c80d4b80736c2fd3ea7a19690ca3498727a8adbdc47791176128a8d64bef0883b888db809f0477355bd00c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_0317e8de5163a14cef0e9ee55c10748596dde21b70124a281b68c11211934fab"}}