{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_eda93c267b0b03452740c6faf2f4547d753073b517a07cbddbae63bd7f6d364c","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_eda93c267b0b03452740c6faf2f4547d753073b517a07cbddbae63bd7f6d364c","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"53b470ad4641c88d1affadb8b96dd73bafbffe53159123b8845de02f4f3b0778","published":"Fri, 24 Jul 2026 00:00:00 -0400","receipt_hash":"53b470ad4641c88d1affadb8b96dd73bafbffe53159123b8845de02f4f3b0778","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"53b470ad4641c88d1affadb8b96dd73bafbffe53159123b8845de02f4f3b0778","observed_at":"2026-07-24T04:43:08.456021Z","parent_run_hash":"b018378f86139a28e6209ec008b31c1282cd1b5c1632dbd43b054c17aa88ab96","published":"Fri, 24 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2510.04019v3 Announce Type: replace-cross \nAbstract: Diffusion large language models (dLLMs) represent a promising alternative to autoregressive LLMs; however, the lack of effective post-training techniques, including reinforcement learning (RL), remains a key challenge for dLLMs, especially for downstream applications. Existing approaches often rely on a sequence-level view that requires biased likelihood approximations. In this work, we propose Amortized Group Relative Policy Optimization (AGRPO), a policy gradient algorithm that leverages the Markovian nature of dLLMs, optimizing individual denoising steps rather than full sequences. Our approach improves alignment between the trained policy and the inference process and also admits efficient, unbiased gradient updates via a novel timestep estimation scheme. We demonstrate AGRPO's effectiveness on different math and reasoning tasks, achieving absolute accuracy gains of +59.4\\% and +69.7\\% on Countdown and Sudoku over the base ","title":"Simple Policy Gradients for Reasoning with Diffusion Language Models","url":"https://arxiv.org/abs/2510.04019","vendor":"arxiv_cs_ai"},"summary":"arXiv:2510.04019v3 Announce Type: replace-cross \nAbstract: Diffusion large language models (dLLMs) represent a promising alternative to autoregressive LLMs; however, the lack of effective post-training techniques, including reinforcement learning (RL), remains a key challenge for dLLMs, especially for downstream applications. Existing approaches often rely on a sequence-level view that requires biased likelihood approximations. In this work, we propose Amortized Group Relative Policy Optimization (AGRPO), a policy gradient algorithm that leverages the Markovian nature of dLLMs, optimizing individual denoising steps rather than full sequences. Our approach improves alignment between the trained policy and the inference process and also admits efficient, unbiased gradient updates via a novel timestep estimation scheme. We demonstrate AGRPO's effectiveness on different math and reasoning tasks, achieving absolute accuracy gains of +59.4\\% and +69.7\\% on Countdown and Sudoku over the base ","title":"Simple Policy Gradients for Reasoning with Diffusion Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-24T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2510.04019"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a2796077e7fafc6da831fcd19f4ca03f73995263498b7b732d16f6004cc16b3c20ca4125b2139f7d642959902993f77e39a02bcde79e9f8f262f1afac9c96e0d","signer":"crovia.substrate","subject":{"observed_at":"2026-07-24T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2510.04019"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"00e850ccc7a80345dd2e9674232a787099d4f94f33dae38bb3fa26488c12ed10","leaf_index":347240,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6c54f194690bc3ec485e85505e4b3f60753b593a5ca7fdd673764ee45cdf7185","side":"right"},{"sibling":"1c4022fd068f37c3b7f46b3b5de1e5df7dfd61abaca83354636339aea485ca7a","side":"right"},{"sibling":"7ab221db7ddcae09786b5f4e614dc2053e4f9c0800f52f2f9086935269d358e2","side":"right"},{"sibling":"c6e718b27918609860c4e210aa2d70f01943e095747b7a3a8fe9b83e16d16d31","side":"left"},{"sibling":"40506af0a083675f981cc82ed25d9eac411c1ca9b5804512c48cd2bbaef63140","side":"right"},{"sibling":"c742a19ac363bd34951e87d604829adff6d3c8dd3daeb979155f1f76fdbb63c7","side":"left"},{"sibling":"22949f5b6c661121ddf62ce35bcf81cd2664e41e1c6aeed348c449bca292ce6b","side":"left"},{"sibling":"aa4b293ba10895bb7f8b28f0f04360b53a8fc5a7523d9a560dbb7b29d003643e","side":"right"},{"sibling":"759f431c97765cb7fb12d38ee64fd6a87076abf1b63c87d7c8c172d57b29084d","side":"right"},{"sibling":"b9cb83ecb59812b3b9270c365951fa79dead288c3923b6cf1f83ffb911b27405","side":"right"},{"sibling":"e6c9083cd0939b38f691f619de2054068654e9c77f7c7ab051e0d48b77339e5d","side":"left"},{"sibling":"d3139af8c5ce235438e1c69e4b7afa44ba09129fd86674968434f23e546f423e","side":"left"},{"sibling":"cb89775a838ee16d10fc8da3213420c2012b4d96e8d55cd49939b0887a4b92d3","side":"right"},{"sibling":"252d30ea8052c3bb6b40bc5cc29fc9b9725343d212f84c08fbbae4215a125b00","side":"right"},{"sibling":"f3e45bceed774d2402fa45d41ff5190f295823bd2f216eb90157884150034693","side":"left"},{"sibling":"3cfa2102c0224815c6f3bf73e6710e24103f43f7bf5da1ca2abad1416d9c0890","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"e871fd7edf9b2ad89bce1609a028f5225eea4d14372169bac242420830f86530","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":347413,"merkle_root":"9efe042c5dd6583dfd3b6a58fbfc289807f60bcf2bd2927f10488a54a8ba11fc","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260724T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-24T05:38:42Z","sig_algorithm":"ed25519","signature":"8633c55f558d42994850505218b2862c6134bad2b1c80d4b80736c2fd3ea7a19690ca3498727a8adbdc47791176128a8d64bef0883b888db809f0477355bd00c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_eda93c267b0b03452740c6faf2f4547d753073b517a07cbddbae63bd7f6d364c"}}