{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b09ed020466f88c64f98b4fd0531524ee6509b58e62c756a34766e7b0506794f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b09ed020466f88c64f98b4fd0531524ee6509b58e62c756a34766e7b0506794f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"30435bbd322c07dc47a7e3e11128c966f544a6ad878db8af677999f174063094","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"30435bbd322c07dc47a7e3e11128c966f544a6ad878db8af677999f174063094","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"30435bbd322c07dc47a7e3e11128c966f544a6ad878db8af677999f174063094","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2507.04136v2 Announce Type: replace \nAbstract: This survey offers a comprehensive foundation on the integration of RL with language models, highlighting prominent algorithms such as Proximal Policy Optimization (PPO), Q-Learning, and Actor-Critic methods. Additionally, it provides an extensive technical overview of RL techniques specifically tailored for LLMs, including foundational methods like Reinforcement Learning from Human Feedback (RLHF) and AI Feedback (RLAIF), as well as advanced strategies such as Direct Preference Optimization (DPO) and Group Relative Policy Optimization (GRPO). We systematically analyze their applications across domains, i.e., from code generation to tool-augmented reasoning. Crucially, we move beyond descriptive categorization to provide a rigorous algorithmic analysis of failure modes, mathematically framing the structural bottlenecks and stability trade-offs inherent in policy optimization. We also present a comparative taxonomy based on reward mod","title":"A Technical Survey of Reinforcement Learning Techniques for Large Language Models","url":"https://arxiv.org/abs/2507.04136","vendor":"arxiv_cs_ai"},"summary":"arXiv:2507.04136v2 Announce Type: replace \nAbstract: This survey offers a comprehensive foundation on the integration of RL with language models, highlighting prominent algorithms such as Proximal Policy Optimization (PPO), Q-Learning, and Actor-Critic methods. Additionally, it provides an extensive technical overview of RL techniques specifically tailored for LLMs, including foundational methods like Reinforcement Learning from Human Feedback (RLHF) and AI Feedback (RLAIF), as well as advanced strategies such as Direct Preference Optimization (DPO) and Group Relative Policy Optimization (GRPO). We systematically analyze their applications across domains, i.e., from code generation to tool-augmented reasoning. Crucially, we move beyond descriptive categorization to provide a rigorous algorithmic analysis of failure modes, mathematically framing the structural bottlenecks and stability trade-offs inherent in policy optimization. We also present a comparative taxonomy based on reward mod","title":"A Technical Survey of Reinforcement Learning Techniques for Large Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2507.04136"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b88eddf5b8021b17609ff6952225f58837fec1b60b91704f6432f737740f9fc693bee1b03b5b418a4dc22e51d4188a1901c61fc7790d9121218fd5062195b70b","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2507.04136"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"e0c74bc2ebd007639a8eb45480c14ba52b759f1eadf1c8f3ef99bb865f881f65","leaf_index":289169,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"bedb9de963bcfb9487e70d8183269f21ecdc948b982def37d149af2859eff869","side":"left"},{"sibling":"ee5699fd990873ea8e172da6e9f84547d068097b0f8550cd0dc2d6d4a9181781","side":"right"},{"sibling":"d464de97ba7965d23484c3452cd2e9125e0c41e02c811764fc5d481b1716121e","side":"right"},{"sibling":"a42679a021c99f3f750339090bec150db25bed78b61e7c86eebbf7cfadbd572b","side":"right"},{"sibling":"44ef19f058dfd8d73cb399bf76855759ad26f52f90b910341addadb0c79086f5","side":"left"},{"sibling":"5d89f664a906e8defe2abbe89c134df0a9bc15697f6b1ab4c1f1a61ad5a75bdf","side":"right"},{"sibling":"cfebf4e758f80b6c0a2520ffba9d9d900a232da3623a4c530a1c712d33cffc24","side":"right"},{"sibling":"a6e115fb6d42f8f126d042702270c371c7df6161120ef9b37366edddc165bd28","side":"left"},{"sibling":"35d3e8088ee93171aa479055dcd001cfb6d23925114ddc0908531e54298b0d29","side":"left"},{"sibling":"841129c21a7583176cdc7de281cadfe0e00d04673461e199760cd5128d8cc2d5","side":"right"},{"sibling":"19d6dfd29bc47f35fa02e8fe765277ba9cc3e6da5072309f24ebaac5b5f295e3","side":"right"},{"sibling":"8e0ad7889eb2d4b40e5b6c3d8e2eb19d4e202374983f468aa76321823de07a9f","side":"left"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b09ed020466f88c64f98b4fd0531524ee6509b58e62c756a34766e7b0506794f"}}