{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_8236a4a686f57e8b7c6447350cdc74435d5d140d11b9485bfb1de60a95f7c6e0","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_8236a4a686f57e8b7c6447350cdc74435d5d140d11b9485bfb1de60a95f7c6e0","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"2fe918f0e78dcf80802cffa83119ea54d952592ccb452d95d14c76b12f9d8401","published":"Thu, 23 Jul 2026 00:00:00 -0400","receipt_hash":"2fe918f0e78dcf80802cffa83119ea54d952592ccb452d95d14c76b12f9d8401","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"2fe918f0e78dcf80802cffa83119ea54d952592ccb452d95d14c76b12f9d8401","observed_at":"2026-07-23T04:43:18.446221Z","parent_run_hash":"b3f5e4095688e31d15c25de2607bca42111343bdfdac967d48e47905415ddea0","published":"Thu, 23 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.19691v1 Announce Type: cross \nAbstract: Reinforcement learning with verifiable rewards has become the predominant recipe for eliciting test-time scaling in explicit Chain-of-Thought reasoners. Yet this scaling path remains computationally costly, since every intermediate step must be decoded as a language token. Latent reasoning instead carries intermediate computation as continuous vectors and already matches or surpasses explicit CoT at far shorter horizons. Despite this promise, latent reasoners remain largely imitation-bound, while explicit CoT has already moved past imitation via outcome-reward RL. Latent trajectories lack a tractable per-step likelihood and an adaptive stopping interface under fixed thinking budgets, so outcome rewards cannot elicit latent test-time scaling. We introduce Surrogate Latent Policy Optimization (SLPO) to bring outcome-reward RL to autoregressive latent reasoners: an empirical surrogate policy density over latent transitions for trajectory-","title":"SLPO: Scaling Latent Reasoning via a Surrogate Policy","url":"https://arxiv.org/abs/2607.19691","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.19691v1 Announce Type: cross \nAbstract: Reinforcement learning with verifiable rewards has become the predominant recipe for eliciting test-time scaling in explicit Chain-of-Thought reasoners. Yet this scaling path remains computationally costly, since every intermediate step must be decoded as a language token. Latent reasoning instead carries intermediate computation as continuous vectors and already matches or surpasses explicit CoT at far shorter horizons. Despite this promise, latent reasoners remain largely imitation-bound, while explicit CoT has already moved past imitation via outcome-reward RL. Latent trajectories lack a tractable per-step likelihood and an adaptive stopping interface under fixed thinking budgets, so outcome rewards cannot elicit latent test-time scaling. We introduce Surrogate Latent Policy Optimization (SLPO) to bring outcome-reward RL to autoregressive latent reasoners: an empirical surrogate policy density over latent transitions for trajectory-","title":"SLPO: Scaling Latent Reasoning via a Surrogate Policy","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-23T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.19691"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a63fc804326cc85975778e8f5ac4ffce74986c67a490cfca2b8d236e4b5174c799823bac5a76f137d60f68662ed77b9d87a61d5db69d13c860a185a9cb113406","signer":"crovia.substrate","subject":{"observed_at":"2026-07-23T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.19691"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c9a9b9f6bce9c749cbc8c3f8273165630d34071e9ac69383b90dc228a685d05b","leaf_index":343683,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"665357b49d5d69b5f0e87ece2bd434e006d780e4e85882b6908918253a1c7e30","side":"left"},{"sibling":"0518fc8a41dc6e369050880cd488d10faa262ffa8bb61f3f6fac8d3ea1839c68","side":"left"},{"sibling":"7b9dd5c0ea46e149803189d1c5756a582f964b4879c35b0c42228aa04ff7f56b","side":"right"},{"sibling":"6308cf2f0e0af3d90dd1038c4e93290c5bd1e88567726988149e9a08bed70811","side":"right"},{"sibling":"e2ebe5a4c0046d068bd71a5efc2e85dc91595eb6634aab443db45868afab9f3c","side":"right"},{"sibling":"7e3984de88de55081321eded10483bf96cbd7dc7a1b4a5477d3a5b864821f293","side":"right"},{"sibling":"ee80ef54fb43556fbc0cdc6b587800c87489c2cca696f33b35a13d71508c0f19","side":"right"},{"sibling":"fbe8bc0f33cd1a297ce2f845b101ed1107ebb67f2c2577c28ed032c1046d3b97","side":"left"},{"sibling":"6ecdcc1e2fb6ab44777fffbb7c8297d723018c63aac9d90c7a1bbebe728d84cf","side":"right"},{"sibling":"93d7d8e0e882d05b2a15bb707a824979a0427907eb47e687c712906674a0d345","side":"left"},{"sibling":"92219a3ef58cd145d94f071b0b9396cec3707812b02a8c7c63f2d0e22340552b","side":"left"},{"sibling":"4eb402d67bd4bf583b0434363061166fe259c34cc6c42adb32dcfbff0a9f5767","side":"left"},{"sibling":"2dd9cb2521044ee7c6b74f2315e0a0253b8df0d04a7b810bbbbe7da5a9788769","side":"left"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"941f71d7ce3990a507b8f485de3f872a51d0e9a8c59405d76c2ae6f4a6af494a","side":"right"},{"sibling":"6281b6f7a93c44e3c4895bc65cfcb6f2be24dd725f4f46540ec022a6e215f4e8","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"6baede22892163664bb2e4d92cf75e6b290761533a6c39491c3afd89bb3a0252","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":343944,"merkle_root":"afab58f71597d393a7a61b857dac2eacb72fd1c04cd1432c2622d7b19309dffd","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260723T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-23T05:38:39Z","sig_algorithm":"ed25519","signature":"1a58fdc6367d0c86f83d2385be748e0800a64bcf9987c92fadca53deda009cd390d2070302c2b6e44841febfd864637c323ba12c7a1e9ae58fdf2a0cf546ac01","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_8236a4a686f57e8b7c6447350cdc74435d5d140d11b9485bfb1de60a95f7c6e0"}}