{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c694c78d3040975873c4fd29e85eb5dba60ec533935ca0aab7afe016b5faa4f1","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c694c78d3040975873c4fd29e85eb5dba60ec533935ca0aab7afe016b5faa4f1","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"004b199faef6febb919190798bc37b0a45b293dd4b914f8e41301e92e84d55a4","published":"Wed, 08 Jul 2026 00:00:00 -0400","receipt_hash":"004b199faef6febb919190798bc37b0a45b293dd4b914f8e41301e92e84d55a4","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"004b199faef6febb919190798bc37b0a45b293dd4b914f8e41301e92e84d55a4","observed_at":"2026-07-08T04:43:57.834712Z","parent_run_hash":"46ab019f0b0f0bfcde5e14ed7c256069c6fa8c8079b9f87fd3a8a6d9259e3864","published":"Wed, 08 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.05541v1 Announce Type: cross \nAbstract: Reinforcement Learning is commonly used to train large language models using environmental feedback. In applied settings, the environment usually provides sparse or delayed feedback. This makes it difficult for the model to pinpoint which actions in its reasoning led to success or failure. So, learning effectively from these signals is hard because the model must determine how each failure should inform meaningful behavioral corrections in subsequent iterations. We introduce a training framework, Self-Review Reinforcement Learning, that embeds an explicit self-review step into each RL episode. When a first-pass response fails, the model generates a self-review to identify what went wrong, which conditions an improved second attempt. Unlike inference-time reflection approaches, such as Reflexion, the framework optimizes self-review with policy gradients and internalizes improvements into the base policy via selective distillation, ensur","title":"Self-Review Reinforcement Learning (SRRL) with Cross-Episode Memory and Policy Distillation","url":"https://arxiv.org/abs/2607.05541","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.05541v1 Announce Type: cross \nAbstract: Reinforcement Learning is commonly used to train large language models using environmental feedback. In applied settings, the environment usually provides sparse or delayed feedback. This makes it difficult for the model to pinpoint which actions in its reasoning led to success or failure. So, learning effectively from these signals is hard because the model must determine how each failure should inform meaningful behavioral corrections in subsequent iterations. We introduce a training framework, Self-Review Reinforcement Learning, that embeds an explicit self-review step into each RL episode. When a first-pass response fails, the model generates a self-review to identify what went wrong, which conditions an improved second attempt. Unlike inference-time reflection approaches, such as Reflexion, the framework optimizes self-review with policy gradients and internalizes improvements into the base policy via selective distillation, ensur","title":"Self-Review Reinforcement Learning (SRRL) with Cross-Episode Memory and Policy Distillation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-08T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.05541"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1e2908f95e3f80552b2cda92083cd5e1e800fa7f8a0150e5da7a716575579eaf8569aa459753746276ca027bad416f372c395114902119e9fc4ce3e88f8ed80d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-08T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.05541"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"ec4e865a1c7f25488691ac41c04768a32cf92f57b935c8e7304dea017ccbc644","leaf_index":292699,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"28f5d5fe861e12bc7e89c9a6bbe103cd8bc2506f19fa09898db350e67660abbb","side":"left"},{"sibling":"a33e9b08a91fb01f3dfe0f59f73019f76ee8dd6b51602bfec13c36dda477502f","side":"left"},{"sibling":"c6e582e2a62946446aa6cf99f1fb0e3db5242853eb5780de4da518f3586320f3","side":"right"},{"sibling":"bc9e3add218eda57801d2b5d528b538d738efaecb7503fbf4d6e5be25ab1f384","side":"left"},{"sibling":"c53e677db9041b39db1da873255bc20bf621f606004a2a4a6a88f8a20a0dd7b7","side":"left"},{"sibling":"72f7f4fe9923986ba7eb993a43289fff6a04211feec32380568db20ee9ddf0ac","side":"right"},{"sibling":"b0c3c5751a1b9461c133083b884a2c1fa5dbc65b069dd53ccc95339ef1af56c5","side":"left"},{"sibling":"7db888cb3645d8b5e3052bd5d092a85490e29ab732a691eb6a8b8560770aa0ff","side":"right"},{"sibling":"1515812edbf9903d3d787f20218b8577e2f1fef32592508ce01a3b3ebe75f507","side":"left"},{"sibling":"d4b475beae71e5b4d5ab7e66f7144e7b8e1356fcd32339e49df234b455659fef","side":"left"},{"sibling":"cead64e0790e8871aeea2334210146b5e35fe7e4bc10748acdf76da0c565be05","side":"left"},{"sibling":"d1231ac6e6bd7d6867a9109fbdadede0b1631e97866ddde70f7ac2d52c28e15f","side":"right"},{"sibling":"76855b4804c75c52bf97aa34358950d42d6103cdc1a86be5f0a2c8de4d65c106","side":"left"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"9e75f2ab0ddf2dc9e92af7049244c21b909734ab57906a35dfad2853ca9966e2","side":"right"},{"sibling":"e6cd4cad39a4b6ca6647d1b0ad2db86e57e5fa6240f65966c10093e91140769b","side":"right"},{"sibling":"90a7efc6b94ec8913fbdf03f4927a821b9fb89921d526716f5ee28f015303779","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":292998,"merkle_root":"f533b7efebdfd8fb6ba3e7cc158ee55261fd53a7985f234cfea359170dad4d5a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260708T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-08T05:38:17Z","sig_algorithm":"ed25519","signature":"07ebb10c732bbffb28b55a5db01d5525f6b8ab7ef36e1e0a68c35c96ded99f7040c277edba75eb6b15c77feda30bd5321e31ada6f572b674d06f4f8e24bd2f07","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c694c78d3040975873c4fd29e85eb5dba60ec533935ca0aab7afe016b5faa4f1"}}