{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_34d3d161b53a571d09cd998b0f42bbac2763c3eaf55a71ec09b99401b48f5f87","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_34d3d161b53a571d09cd998b0f42bbac2763c3eaf55a71ec09b99401b48f5f87","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8091e878cd2c24f2d37fd80fd7f5f8f96fc1ae05f23e1eeb4022bbad80015909","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"8091e878cd2c24f2d37fd80fd7f5f8f96fc1ae05f23e1eeb4022bbad80015909","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8091e878cd2c24f2d37fd80fd7f5f8f96fc1ae05f23e1eeb4022bbad80015909","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.03478v1 Announce Type: new \nAbstract: Post-training of frontier language models is conducted on curated task suites, and inevitably leaves a distribution shift between training and deployment environments. This exposes developers to generalization failures, which are relatively poorly understood. To better understand such generalization failures, we believe the community should construct clean demonstrations under simplified conditions. To facilitate this, we propose a simple and flexible way to construct language models which fail to generalize in controllable ways when subsequently trained with Reinforcement Learning (RL) on a given distribution of training tasks. Our construction uses Supervised Fine-Tuning on a dataset of a mixture of transcripts corresponding to a collection of 'conditional policies', which can each independently be assigned certain behaviors on each different task distribution, to obtain a model that is then well approximated as a 'mixture of condition","title":"Demonstrating Generalization Failures via Mixtures of Conditional Policies","url":"https://arxiv.org/abs/2607.03478","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.03478v1 Announce Type: new \nAbstract: Post-training of frontier language models is conducted on curated task suites, and inevitably leaves a distribution shift between training and deployment environments. This exposes developers to generalization failures, which are relatively poorly understood. To better understand such generalization failures, we believe the community should construct clean demonstrations under simplified conditions. To facilitate this, we propose a simple and flexible way to construct language models which fail to generalize in controllable ways when subsequently trained with Reinforcement Learning (RL) on a given distribution of training tasks. Our construction uses Supervised Fine-Tuning on a dataset of a mixture of transcripts corresponding to a collection of 'conditional policies', which can each independently be assigned certain behaviors on each different task distribution, to obtain a model that is then well approximated as a 'mixture of condition","title":"Demonstrating Generalization Failures via Mixtures of Conditional Policies","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.03478"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:cfa9c42fee276fada51a9e6521ef68349c978005f7fbfe5da382eb55d4c3048cec7a9862d58475b37041fa1f8c9024ea5e533145ae24321112324a2d7e63ac00","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.03478"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"00720e066f8502ef848e9fde969c97f64149db5bf2fe0c671acc0ae2e0757ead","leaf_index":288734,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"43a8ceda3ac86772f35abbd2de8b5432f8b4853bf5a52c2198569be51b910a83","side":"right"},{"sibling":"e5f82af6494a431e286e4eefd70397f0d3c637e3741062352ca5dc9a5cb8539d","side":"left"},{"sibling":"a5e40dda153776f5ee2369e325ec0c698f09b859a1c98afa78420ab2ce095ed6","side":"left"},{"sibling":"6c1fec04e034a25cfdfea9291609dd5190f2af28d4b85a36a873d3a1e1ea03ad","side":"left"},{"sibling":"31347ca06a4774b8956440f8193fa1632894bf12e556e7082239a53fb87cd6ce","side":"left"},{"sibling":"c76dc76bc3b8846cbe60652cbb6f294981e01febd0141a65773df572e2064658","side":"right"},{"sibling":"61f7272108de819a7a77493153b988651785961143dd7f7cbf21c3d662725b0d","side":"left"},{"sibling":"693d22f9477576140ca2120520ac7b4b633fb122e7f590d3abc01e28b77f9cfe","side":"left"},{"sibling":"2b24ae0b0e86a6bb85711be0b56eabcfea8026b8cad7364829595d5562ea5b79","side":"left"},{"sibling":"66acb8614a0400fe91b4430bfa4ca8f7f359fb95aea361dfea6a7c88a9fe24a7","side":"left"},{"sibling":"b76af82f0e95812185e25016354f4707e41c1bacfe819e4948e5e364745511e8","side":"left"},{"sibling":"175b61fd9088baa970ad449ad7fc5d7babb21d38120cfa8c28053ae9d448ac83","side":"right"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_34d3d161b53a571d09cd998b0f42bbac2763c3eaf55a71ec09b99401b48f5f87"}}