{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_387d8d1b37afcbd74513e70f607f8a9fb81f65a53c8c0ca65145fddc8f7e404b","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_387d8d1b37afcbd74513e70f607f8a9fb81f65a53c8c0ca65145fddc8f7e404b","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"3a7cfd50b3ee258d699b2b5d7882f99606a921a2d6132ef94377452f0c8d9b37","published":"Wed, 20 May 2026 00:00:00 -0400","receipt_hash":"3a7cfd50b3ee258d699b2b5d7882f99606a921a2d6132ef94377452f0c8d9b37","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"3a7cfd50b3ee258d699b2b5d7882f99606a921a2d6132ef94377452f0c8d9b37","observed_at":"2026-05-20T04:43:44.562035Z","parent_run_hash":"5f904c2c2fecb6b44f2adce8bdc9de914b9a39f7c4fdd7bf086a6c2361a30f8c","published":"Wed, 20 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2603.29501v2 Announce Type: replace-cross \nAbstract: Many value-based deep reinforcement learning algorithms rely on target networks - lagged copies of the online network - to stabilize training. While effective, this mechanism introduces a fundamental stability-recency tradeoff: slower target updates improve stability but reduce the recency of learning signals, hindering convergence speed. We propose Target-Aligned Reinforcement Learning (TARL), a simple drop-in refinement for existing algorithms that emphasizes transitions for which the target and online network estimates are highly aligned. By focusing updates on well-aligned targets, TARL mitigates the adverse effects of stale target estimates while retaining the stabilizing benefits of target networks. We empirically demonstrate consistent improvements within discrete and continuous control algorithms across various benchmark environments without any hyperparameter tuning, including a 38.18% peak score gain on Atari-10, whil","title":"Target-Aligned Reinforcement Learning","url":"https://arxiv.org/abs/2603.29501","vendor":"arxiv_cs_ai"},"summary":"arXiv:2603.29501v2 Announce Type: replace-cross \nAbstract: Many value-based deep reinforcement learning algorithms rely on target networks - lagged copies of the online network - to stabilize training. While effective, this mechanism introduces a fundamental stability-recency tradeoff: slower target updates improve stability but reduce the recency of learning signals, hindering convergence speed. We propose Target-Aligned Reinforcement Learning (TARL), a simple drop-in refinement for existing algorithms that emphasizes transitions for which the target and online network estimates are highly aligned. By focusing updates on well-aligned targets, TARL mitigates the adverse effects of stale target estimates while retaining the stabilizing benefits of target networks. We empirically demonstrate consistent improvements within discrete and continuous control algorithms across various benchmark environments without any hyperparameter tuning, including a 38.18% peak score gain on Atari-10, whil","title":"Target-Aligned Reinforcement Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-20T04:43:44Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2603.29501"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:934ba87ebc496a0e228c431f6b7fda9ca569c095a2a1418f43813593ae2f0c333c297c370eac071985d419615b3570ec27509c6430c57c2621639518873b7b05","signer":"crovia.substrate","subject":{"observed_at":"2026-05-20T04:43:44Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2603.29501"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"36b796c4e9f6e47747c0885b4c199eda911d75ec0ea2df119a8d630cf7bf22a3","leaf_index":145302,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"16115434162256ac690e3e8107bb6d2e711c50ed1ac7f13d502f8e82d3dbb43c","side":"right"},{"sibling":"d1d3ea4ae49782d9e5c1b6f4f8b2a292cc232ca4c1394baec691eb84e4d326ae","side":"left"},{"sibling":"bf0a51868d6e6e7c21d8ad0f8e3b28272df0f2dc0e452f9e24677b5728cd0275","side":"left"},{"sibling":"82e1ee0d120f202bd1d1d408099667770bf21a92b38e057758be1da9be815275","side":"right"},{"sibling":"b5927837a88c29fcb532c073b94ec23f7eb31ca0b9918d31caff120a3df73547","side":"left"},{"sibling":"01ed1e09ca55b8947e9e2a49f489627f1a042f0c03eac8fb97fed38389096fa5","side":"right"},{"sibling":"cefd93338369caa43fee7488c87e004962f685c84a7eb2dabfe358e69b38d6f1","side":"right"},{"sibling":"cc6b64c1eb047b460e9fe9beb452cf69fe3eb490dcf6d93d8193dda021cdf8ef","side":"left"},{"sibling":"f073aa7be27ee7e1d9eb0de5f129f7e9bae0c584fb959378c395b65831dc1d1d","side":"left"},{"sibling":"1526885f19d1fadf6955cf519dbc4e62d593a4bba99d741e8c301740a7068233","side":"left"},{"sibling":"e3a7d5c07f161682d61bd453ffc02ecdf87cfeda70f986d6650017f9d2d6b265","side":"left"},{"sibling":"edbc49f08e5b92291934c05c9e6efd270a6b0698d8d2fa474006366027dfe098","side":"right"},{"sibling":"3e4df6e7457cecbf36f350375e72dcab336a3984422e4c406ef809e4e2944e96","side":"left"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"be08fedc4e72a6fb56606f66812fae7317e09690b9acb18385f4ab117a981238","side":"right"},{"sibling":"0534329a7475dc9df51998c83c16892126679dade0fa34182f21e869599386c7","side":"right"},{"sibling":"2d24720928ead0e7670650eb55f558c4f20e4c18df376f47ba72cfa8cf0ed344","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":147301,"merkle_root":"08903d7159c3b38eeeeafc09eab15139ea417f1d94f02f1fbc87296b37db840a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260521T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-21T18:37:33Z","sig_algorithm":"ed25519","signature":"905f2924632dfa2970c8690285f5b5d4a1d891d0e0ef1cbc404ebec2fd937215ac768e16f0a9f28b18977a55ae0bfd226db6833ae7ef588729054117d2da7303","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_387d8d1b37afcbd74513e70f607f8a9fb81f65a53c8c0ca65145fddc8f7e404b"}}