{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_17a8e96a340fbdd4ba2fb8e5ccd88b665fc8edfd3ea0001fb68861b23b147bc7","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_17a8e96a340fbdd4ba2fb8e5ccd88b665fc8edfd3ea0001fb68861b23b147bc7","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a22981694c01da3177fabdbe1aaadc5e1cf114d0f919fd64c93be365ad10369f","published":"Fri, 29 May 2026 00:00:00 -0400","receipt_hash":"a22981694c01da3177fabdbe1aaadc5e1cf114d0f919fd64c93be365ad10369f","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a22981694c01da3177fabdbe1aaadc5e1cf114d0f919fd64c93be365ad10369f","observed_at":"2026-05-29T04:43:58.478092Z","parent_run_hash":"0fcd87efcfe67ccb9952f747541debc16793919a4d20fd71ca0ad5516a0a13ee","published":"Fri, 29 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2509.21154v4 Announce Type: replace-cross \nAbstract: Process reward models (PRMs) allow for fine-grained credit assignment in reinforcement learning (RL), and seemingly contrast with outcome reward models (ORMs), which assign a single reward to an entire trajectory. However, we provide theoretical proof in this work that the Group Relative Policy Optimization (GRPO) RL algorithm equipped with an ORM is in fact equivalent to a PRM-aware RL objective equipped with a non-trivial, Monte-Carlo-based PRM (given mild assumptions). Leveraging the framework of GRPO-as-a-PRM, we identify a flaw in the GRPO objective that interacts with imbalanced process steps and rewards to hinder both exploration and exploitation (under different conditions). We propose a simple modification to the algorithm to mitigate this defect ($\\lambda$-GRPO), and show that LLMs tuned with $\\lambda$-GRPO outperform LLMs tuned with standard GRPO on downstream reasoning tasks\\textemdash and reach peak performance mor","title":"GRPO is Secretly a Process Reward Model","url":"https://arxiv.org/abs/2509.21154","vendor":"arxiv_cs_ai"},"summary":"arXiv:2509.21154v4 Announce Type: replace-cross \nAbstract: Process reward models (PRMs) allow for fine-grained credit assignment in reinforcement learning (RL), and seemingly contrast with outcome reward models (ORMs), which assign a single reward to an entire trajectory. However, we provide theoretical proof in this work that the Group Relative Policy Optimization (GRPO) RL algorithm equipped with an ORM is in fact equivalent to a PRM-aware RL objective equipped with a non-trivial, Monte-Carlo-based PRM (given mild assumptions). Leveraging the framework of GRPO-as-a-PRM, we identify a flaw in the GRPO objective that interacts with imbalanced process steps and rewards to hinder both exploration and exploitation (under different conditions). We propose a simple modification to the algorithm to mitigate this defect ($\\lambda$-GRPO), and show that LLMs tuned with $\\lambda$-GRPO outperform LLMs tuned with standard GRPO on downstream reasoning tasks\\textemdash and reach peak performance mor","title":"GRPO is Secretly a Process Reward Model","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-29T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2509.21154"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:33c02284c26d5900e9571586eb0c44a67cbcdd63119d6a94b313c43e3b483d88a40ccab846a367e676568f7fe9b572ecd7542996afff92af7f7104a90a007006","signer":"crovia.substrate","subject":{"observed_at":"2026-05-29T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2509.21154"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"efbcbe5d3f1b88b0918a7e7daac3c241bdf20b2ff9a73e04febd38040a82f2cc","leaf_index":158082,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"cabbedfe577d62582dd494e3b16e9501175c8bcad1c70e938c025a32eb348ee5","side":"right"},{"sibling":"b7bfee1d574f8bae4c1fe5e80f1853098f592f4f72f8bdbf7be9ccd14fc0b80f","side":"left"},{"sibling":"427ff3028baaee6535694a475b8e1b4671833431496b9368ed82eba74ab73bb1","side":"right"},{"sibling":"f891d45fa0dd51b1eb95ab477ca1665e6d1a89ebd03caeb20832e2c8f4001cb1","side":"right"},{"sibling":"1db6ee1b844c693165ccc6c3864ccc34458ec09c391bf035931731913117bd52","side":"right"},{"sibling":"796de9dd6ef901ab3e49311279a21042b8f2cfbb09b2fbb767544159218def9b","side":"right"},{"sibling":"62988561fabfe452383ec872bd31f27e80fe019c13c67e0ef3aed8f206215007","side":"right"},{"sibling":"463d4726f68bda2bbf724c72553fd53d88ea99a3d4a713f9b8a1b2d9fde12933","side":"left"},{"sibling":"002471ff5893cd2a2890119cbfe4a9456bfb46cf31fa6aee377650afe6ae0c92","side":"left"},{"sibling":"1dcc44e23fbb0218b13591e4b584eca3600dcf365769cb741e0ecd33b25b8c56","side":"right"},{"sibling":"fcf16a6f44025801f5b83e928acce764352ddbc06ae3043d8cde9a409933e6d8","side":"right"},{"sibling":"78982294dee68f9db9288c64d7e507c7865fda96e1b6f7ccff5c8bb152e93c49","side":"left"},{"sibling":"995b421824624a8282c7f44e64c64ee35344800f477ae1845b41be14d3fab94c","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"eef0e8906a749d3470f89beeedc723f37a5737010bbb0dcc7cf91515338e5a3e","side":"right"},{"sibling":"1a07e481a9407d71aad078ce854cdeee362163c887fe10f889b0ecf0b5e749ad","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":158251,"merkle_root":"485e6b31fe60c8beba5b394808c7e4c32448b2ff65c2482c480ca0e2a2eda718","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260529T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-29T05:37:37Z","sig_algorithm":"ed25519","signature":"bbf9f005201182fce4f9d94c7a9d01a508b56daf7d9611bd73514f5f616bc059d0e5e1f2edfc95716e6fe08ef5fae38b558cbf7f2fd8f9d5dfe4c34a54c83005","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_17a8e96a340fbdd4ba2fb8e5ccd88b665fc8edfd3ea0001fb68861b23b147bc7"}}