{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_8ad721f5bc352388b523a3937bf4fd04d538421796917c0baad3fc2f87ed4bf1","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_8ad721f5bc352388b523a3937bf4fd04d538421796917c0baad3fc2f87ed4bf1","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"7ef50dcc75eab8c1fcd792cae4ca9a2a5c14d59e494d26f52cee41310c613f9e","published":"Mon, 25 May 2026 00:00:00 -0400","receipt_hash":"7ef50dcc75eab8c1fcd792cae4ca9a2a5c14d59e494d26f52cee41310c613f9e","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"7ef50dcc75eab8c1fcd792cae4ca9a2a5c14d59e494d26f52cee41310c613f9e","observed_at":"2026-05-25T04:43:48.841018Z","parent_run_hash":"56713422f06ad87427cd8cdcdb1ed341feb016b33c37198d9b168328d1df15fb","published":"Mon, 25 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.23565v1 Announce Type: cross \nAbstract: Reinforcement learning agents often exhibit unintended goal-directed behaviour outside their training distribution, but we currently lack a principled understanding of how such agents will generalise to novel environments based on their training history. We address this gap for agents trained sequentially on one or more tasks. We study over 100 sequential training pipelines, evaluating behaviour across over 250 out-of-distribution environments. We find that salient features drive generalisation, and that goals learnt early in training can persist and influence those acquired later. To explain these phenomena, we introduce latent policy gradients, a method that predicts what out-of-distribution behaviour a training pipeline will likely induce. Our method simulates the evolution of low-dimensional latent variables during training according to what would achieve high reward on the training objective with respect to a simple model of how t","title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","url":"https://arxiv.org/abs/2605.23565","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.23565v1 Announce Type: cross \nAbstract: Reinforcement learning agents often exhibit unintended goal-directed behaviour outside their training distribution, but we currently lack a principled understanding of how such agents will generalise to novel environments based on their training history. We address this gap for agents trained sequentially on one or more tasks. We study over 100 sequential training pipelines, evaluating behaviour across over 250 out-of-distribution environments. We find that salient features drive generalisation, and that goals learnt early in training can persist and influence those acquired later. To explain these phenomena, we introduce latent policy gradients, a method that predicts what out-of-distribution behaviour a training pipeline will likely induce. Our method simulates the evolution of low-dimensional latent variables during training according to what would achieve high reward on the training objective with respect to a simple model of how t","title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-25T04:43:48Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.23565"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a1209f94f2cdf2799761bf7a6fe2f82400841e8dc56fe0446a34420e97775e81dae77528ddeb8ecdfab8422715b361cd1c15a8ddc2dc741dd5c11aebeb48180f","signer":"crovia.substrate","subject":{"observed_at":"2026-05-25T04:43:48Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.23565"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"067756a675217f952a86df05e1e912e2390ebde7d305e02605d745fe7e6b2068","leaf_index":149637,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1434c38e3c9f2a4c4c57deff2adca48e6520d88ad035d30c15da4c17cffb7086","side":"left"},{"sibling":"3ad6f92fb6c17c357d92bbd699573ee32d4b48446b79d6f1e287cab95a20c1a4","side":"right"},{"sibling":"38b73282cc07fd412a7b6a06ad3d8e56a2ef91ecd9731a270d6223a9f17f9085","side":"left"},{"sibling":"c2ebec8e6c30f15e5216195a3d944d55626a8b0bba9ce2f77f4c455defe08655","side":"right"},{"sibling":"b887423e9ec646b26f77fd3c1d7ebfb950af1b730936f447874e71a25fa38577","side":"right"},{"sibling":"f61fe7709668d9c426e63af2ee198fb571873853484d18d2a5397d0eba756cc2","side":"right"},{"sibling":"919f4560dbaa8584ff90fc1ba33dc74b59dff462d40837f8f3cbb7a4f3d2f375","side":"right"},{"sibling":"cb71a04113e54ec457f452a34b2778696e1747afe98ba9da255ce1f5fd8619d7","side":"left"},{"sibling":"e044acc52c31efe134c902c984299ea3452b42df993d096ed61ad8be9d530bbc","side":"right"},{"sibling":"6be461ecf12dedf98a31921aa7b5c32d4a32df6897e0f697b8bf1a4ba3e2d324","side":"right"},{"sibling":"c2861a8cef3eb66bf2726aa377c24a6bf6e8b7489dcd0e870b61d62a35ccadfb","side":"right"},{"sibling":"134949308b15cffd6792ee2cf678119af34d69a64764d7c89cd47573c94e1cda","side":"left"},{"sibling":"debbc1a6232ea7970009b91bdc2345041b2de5ac59551f955855d579001e512c","side":"right"},{"sibling":"216869846f40bd905626884f58cb67b9019e488b3656946a7808c436ab7339ac","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"e734b0c6d6d13acb5f4cf00367b3913e5bbf1aa717377aa4e4820c6de24b7673","side":"right"},{"sibling":"4bbb7f78e96179bb9cdd06d7207b66b3503ab68e4880438263025b04ebe8f7ec","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":149835,"merkle_root":"51484a548bc0cf3d178eec868e98935c87f0aa83369143b7c96b048585e5ba01","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260525T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-25T05:37:33Z","sig_algorithm":"ed25519","signature":"99a657e53a015ace0bade5d7fbb3f60f950b6d1e1128251a63afcc454a492422cae5efdc0c18503b4f7e62f371bd48f2889be78d9048c4681d6c1e73f754d705","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_8ad721f5bc352388b523a3937bf4fd04d538421796917c0baad3fc2f87ed4bf1"}}