{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_ee8a0c8e944b1e386f126c40c1ff5fc8ec4487eeb292705867d6142e91ea8e42","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_ee8a0c8e944b1e386f126c40c1ff5fc8ec4487eeb292705867d6142e91ea8e42","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"ac6c96ec95a3b2d000653515b0e9d5bb33bde093992c310e9c24f4c165cf33d6","published":"Tue, 28 Jul 2026 00:00:00 -0400","receipt_hash":"ac6c96ec95a3b2d000653515b0e9d5bb33bde093992c310e9c24f4c165cf33d6","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"ac6c96ec95a3b2d000653515b0e9d5bb33bde093992c310e9c24f4c165cf33d6","observed_at":"2026-07-28T04:43:08.282317Z","parent_run_hash":"23a1ef85134515049ced29518443d084afc46fd7c967741e6c6acdbdbbf29939","published":"Tue, 28 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.19691v2 Announce Type: replace-cross \nAbstract: Reinforcement learning with verifiable rewards has become the predominant recipe for eliciting test-time scaling in explicit Chain-of-Thought reasoners. Yet this scaling path remains computationally costly, since every intermediate step must be decoded as a language token. Latent reasoning instead carries intermediate computation as continuous vectors and already matches or surpasses explicit CoT at far shorter horizons. Despite this promise, latent reasoners remain largely imitation-bound, while explicit CoT has already moved past imitation via outcome-reward RL. Latent trajectories lack a tractable per-step likelihood and an adaptive stopping interface under fixed thinking budgets, so outcome rewards cannot elicit latent test-time scaling. We introduce Surrogate Latent Policy Optimization (SLPO) to bring outcome-reward RL to autoregressive latent reasoners: an empirical surrogate policy density over latent transitions for tra","title":"SLPO: Scaling Latent Reasoning via a Surrogate Policy","url":"https://arxiv.org/abs/2607.19691","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.19691v2 Announce Type: replace-cross \nAbstract: Reinforcement learning with verifiable rewards has become the predominant recipe for eliciting test-time scaling in explicit Chain-of-Thought reasoners. Yet this scaling path remains computationally costly, since every intermediate step must be decoded as a language token. Latent reasoning instead carries intermediate computation as continuous vectors and already matches or surpasses explicit CoT at far shorter horizons. Despite this promise, latent reasoners remain largely imitation-bound, while explicit CoT has already moved past imitation via outcome-reward RL. Latent trajectories lack a tractable per-step likelihood and an adaptive stopping interface under fixed thinking budgets, so outcome rewards cannot elicit latent test-time scaling. We introduce Surrogate Latent Policy Optimization (SLPO) to bring outcome-reward RL to autoregressive latent reasoners: an empirical surrogate policy density over latent transitions for tra","title":"SLPO: Scaling Latent Reasoning via a Surrogate Policy","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-28T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.19691"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:e5571af48c6144e4eaeb9e5c65a2279f532a13e42cee4e63cc361a387b25487b1629b5d04d23e4f06cb7ab91410624d7f3bc24ec2c77f9750e954f7f7a3ec208","signer":"crovia.substrate","subject":{"observed_at":"2026-07-28T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.19691"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"e2a0d1a0dde330a9e9f6614bb0fe84f8dc9e127131bb59207db766d95412a9d0","leaf_index":360884,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"cbe47d21575daa9d055ab4c82f4fe78742307a398635d0d9a0c8b973c0e9edbf","side":"right"},{"sibling":"1c883824cfc4b5e1dae5309c63f7d663dd9cae057375c3d1a0c6f1c471899d1b","side":"right"},{"sibling":"c8fa7eb7a7602f43b656b9869a8bb1978a626936a4bb9c2eb6b5201af5cb5aa8","side":"left"},{"sibling":"16ddc091e37ef9457b8a5a59270b51bb511649fd6e10de418b42ea7caa6892de","side":"right"},{"sibling":"6539680124be497b4c1d4d6b33ed78568fb4d9c97e93c918777cf25b6f431c86","side":"left"},{"sibling":"70cf2b76d17391f6de2eacf64c4eb257ef2ab322e8314ec2f60a5dbc5c0a78d4","side":"left"},{"sibling":"3624810eba28b73089ab25eb96ab3b98f4fccbf630cdc8ce257ab18ffa11f60a","side":"right"},{"sibling":"2269d88b8ead175347e1333136f38876970c71eeededc634500c20d221068882","side":"left"},{"sibling":"f05fa4d2088913dc5410109b6c9f2c28b5e413f06a98f46cd68973c567c3b416","side":"left"},{"sibling":"fe03c6d0b083c4097049d7fc6d7d08dfdb05d9c198b2786336689712d66eabf0","side":"right"},{"sibling":"590ea76fdfc1b9e8072055c378be3182f916709e8d06bd955232e650a7188b86","side":"right"},{"sibling":"d92781c59301ffd5bfb0bad75d9fdf6d73879f715518149d383f7362213e0daf","side":"right"},{"sibling":"f315303d4402b57497416d48eb4c4bb50405b40862d41c7caf318cb3d29c5237","side":"right"},{"sibling":"36973eb5f586cd67e0c0dc055dd87e734aa544c35d4400d2c9f932f8ef8fb27f","side":"right"},{"sibling":"e39f7900355489c4718b21cc2d3d06382e1d2f22b864d10be9ebc9d498b279a1","side":"right"},{"sibling":"1f9a970b25dd938c98cabc9e0a55c5a6f46292b9fa37cc89d90ef0cbb1e05a8c","side":"left"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"3b50864499c874394ea0928567747666eaf59b01380e46cd52164ec5acec0f71","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":361008,"merkle_root":"3065e8369ea437c06beba806dc4e4bb159979adeb21fe632242c1906a7204647","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260728T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-28T05:38:48Z","sig_algorithm":"ed25519","signature":"9141644407577a82611c1579110f667de2d46dc6b93c6322edf26f4c3056ea99f0e56502853908e30d87c38bcf95eb6e0ab5130525aa51505bd6f61938120609","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_ee8a0c8e944b1e386f126c40c1ff5fc8ec4487eeb292705867d6142e91ea8e42"}}