{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1beba5c109c40a59fb2921695c97378cabe1de03a1eb230b31a5695653ef4e1a","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1beba5c109c40a59fb2921695c97378cabe1de03a1eb230b31a5695653ef4e1a","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8ffc3ed5b3a9ff8440d38b18a33072b9d43d257bfecc3fd76309e7770523ca76","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"8ffc3ed5b3a9ff8440d38b18a33072b9d43d257bfecc3fd76309e7770523ca76","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8ffc3ed5b3a9ff8440d38b18a33072b9d43d257bfecc3fd76309e7770523ca76","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.04242v1 Announce Type: new \nAbstract: Group-based reinforcement learning (RL) has become an effective paradigm for improving large language model agents on long-horizon interactive tasks. To obtain finer-grained policy updates than trajectory-level optimization, recent work has moved toward step-level group-based RL, where intermediate steps are grouped and compared within a rollout batch. However, step-level advantage estimation is sensitive to how groups are formed: grouping by broad state keys improves coverage but may compare actions taken under different histories, while enforcing historical consistency yields fairer comparisons at the cost of fragmented groups and missing peer-comparison signal. In this paper, we propose ProGPO (Progress- and Reliability-Oriented Group Policy Optimization), a learned-critic-free method for context-consistent step-level learning. ProGPO keeps exact-prefix action comparison, and complements sparse peer comparisons with transition credit ","title":"Progress- and Reliability-Oriented Group Policy Optimization for Agentic Reinforcement Learning","url":"https://arxiv.org/abs/2607.04242","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.04242v1 Announce Type: new \nAbstract: Group-based reinforcement learning (RL) has become an effective paradigm for improving large language model agents on long-horizon interactive tasks. To obtain finer-grained policy updates than trajectory-level optimization, recent work has moved toward step-level group-based RL, where intermediate steps are grouped and compared within a rollout batch. However, step-level advantage estimation is sensitive to how groups are formed: grouping by broad state keys improves coverage but may compare actions taken under different histories, while enforcing historical consistency yields fairer comparisons at the cost of fragmented groups and missing peer-comparison signal. In this paper, we propose ProGPO (Progress- and Reliability-Oriented Group Policy Optimization), a learned-critic-free method for context-consistent step-level learning. ProGPO keeps exact-prefix action comparison, and complements sparse peer comparisons with transition credit ","title":"Progress- and Reliability-Oriented Group Policy Optimization for Agentic Reinforcement Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.04242"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:002159bab094b54fb2415ea7beb5313c3457025643f52256eca47ddea5b853edaeeddef4691572b388e71e4f223a1cd4c82138e1d475eb1ced234888aab64f0b","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.04242"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c1c3337a9e81a0c47bd39c755fac1249c87318c0041e46aad16739431516e4fa","leaf_index":288757,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"3e863ce49141450aa8145550fb8b3a2000e9aec7d60ff12d021937a0a93716ff","side":"left"},{"sibling":"9e40b2334126f9cf87ad0be18c9c9f8a702d46a61b1fd1c12a8082d7c374d272","side":"right"},{"sibling":"5467c8ae3a7f31a3f89adfeae204e9a4313ca02e55065984d4ba650b65ce01e3","side":"left"},{"sibling":"8adee33e45b2e4c76aa4548ba9992f78ade8310855586d84faf461f1312a0887","side":"right"},{"sibling":"5547927da08df8599cc06965777697fcd81850beeacc9c257378b724de456bea","side":"left"},{"sibling":"6ce2490462f2183557d667ea543ad19b275e012200585fa65831bfe675408ade","side":"left"},{"sibling":"61f7272108de819a7a77493153b988651785961143dd7f7cbf21c3d662725b0d","side":"left"},{"sibling":"693d22f9477576140ca2120520ac7b4b633fb122e7f590d3abc01e28b77f9cfe","side":"left"},{"sibling":"2b24ae0b0e86a6bb85711be0b56eabcfea8026b8cad7364829595d5562ea5b79","side":"left"},{"sibling":"66acb8614a0400fe91b4430bfa4ca8f7f359fb95aea361dfea6a7c88a9fe24a7","side":"left"},{"sibling":"b76af82f0e95812185e25016354f4707e41c1bacfe819e4948e5e364745511e8","side":"left"},{"sibling":"175b61fd9088baa970ad449ad7fc5d7babb21d38120cfa8c28053ae9d448ac83","side":"right"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1beba5c109c40a59fb2921695c97378cabe1de03a1eb230b31a5695653ef4e1a"}}