{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_7d1d8b94ceef385b6ed62b169726d78400a583bf904d4ab089bdcee09e27ccbf","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_7d1d8b94ceef385b6ed62b169726d78400a583bf904d4ab089bdcee09e27ccbf","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"be1d108d75a1c34cc44ace715812bf95ba679474f625954b50a63257f7dc2e94","published":"Tue, 28 Jul 2026 00:00:00 -0400","receipt_hash":"be1d108d75a1c34cc44ace715812bf95ba679474f625954b50a63257f7dc2e94","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"be1d108d75a1c34cc44ace715812bf95ba679474f625954b50a63257f7dc2e94","observed_at":"2026-07-28T04:43:08.282317Z","parent_run_hash":"23a1ef85134515049ced29518443d084afc46fd7c967741e6c6acdbdbbf29939","published":"Tue, 28 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.05547v2 Announce Type: replace-cross \nAbstract: RL-based post-training with GRPO is widely used to improve large language models on individual reasoning tasks. However, real-world deployment requires reliable performance across diverse tasks. A straightforward multi-task adaptation of GRPO often leads to imbalanced outcomes, with some tasks dominating optimization while others stagnate. Moreover, tasks can vary widely in how frequently prompts yield zero advantages (and thus zero gradients), which further distorts their effective contribution to the optimization signal. To address these issues, we propose a novel Multi-Task GRPO (MT-GRPO) algorithm that (i) dynamically adapts task weights to explicitly optimize worst-task performance and promote balanced progress across tasks, and (ii) introduces a ratio-preserving sampler to ensure task-wise policy gradients reflect the adapted weights. Experiments on both 3-task and 9-task settings show that MT-GRPO consistently outperform","title":"Multi-Task GRPO: Reliable LLM Reasoning Across Tasks","url":"https://arxiv.org/abs/2602.05547","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.05547v2 Announce Type: replace-cross \nAbstract: RL-based post-training with GRPO is widely used to improve large language models on individual reasoning tasks. However, real-world deployment requires reliable performance across diverse tasks. A straightforward multi-task adaptation of GRPO often leads to imbalanced outcomes, with some tasks dominating optimization while others stagnate. Moreover, tasks can vary widely in how frequently prompts yield zero advantages (and thus zero gradients), which further distorts their effective contribution to the optimization signal. To address these issues, we propose a novel Multi-Task GRPO (MT-GRPO) algorithm that (i) dynamically adapts task weights to explicitly optimize worst-task performance and promote balanced progress across tasks, and (ii) introduces a ratio-preserving sampler to ensure task-wise policy gradients reflect the adapted weights. Experiments on both 3-task and 9-task settings show that MT-GRPO consistently outperform","title":"Multi-Task GRPO: Reliable LLM Reasoning Across Tasks","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-28T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.05547"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:59ba653863b06ecf371b2f791f15720c1e217723c1290e2bbe0aea4e0bc243749b9c617b2d7bd7e05cef46c9e208c19cb16caa760c1c2c117060869b1cdeb300","signer":"crovia.substrate","subject":{"observed_at":"2026-07-28T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.05547"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"05e62fc1ed9c034df7185cb204c05ae2e46991f1f650ba56dddfb691fd52484e","leaf_index":360810,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"15634900366c2f10621fb61f34ee4f4815eff84e337a0b70d862c1e5e7a8c829","side":"right"},{"sibling":"078f0132c14dede07946fe1b8ac3c4fdd19a22d1cf1076106177bb5f90df8d83","side":"left"},{"sibling":"4565828f3c7a5dc9b195d1f0641bc1962a44544b1641bf904d5f55528d8a1b07","side":"right"},{"sibling":"e59d71a4dbd58e1f53b1f9d7b30867f24794f7ad1ae7ed0bd9ec76210ee0e6de","side":"left"},{"sibling":"c89a939fa4dd22bd7362c6f0106e4f6a3054d783e26c042396cf85753d155340","side":"right"},{"sibling":"e8670decef930025bf646e3a6631cf649dfa52d095f30e0bdf66e5880e39a55e","side":"left"},{"sibling":"edee58372a639bd2a7ffe555c89a5f76b628b6fb4ff60a0320a74d44c214bdd6","side":"left"},{"sibling":"80654268ff95f1daa72785465b969eee9ea2b22da6f8dfc0d2cf6148d4718a52","side":"right"},{"sibling":"f05fa4d2088913dc5410109b6c9f2c28b5e413f06a98f46cd68973c567c3b416","side":"left"},{"sibling":"fe03c6d0b083c4097049d7fc6d7d08dfdb05d9c198b2786336689712d66eabf0","side":"right"},{"sibling":"590ea76fdfc1b9e8072055c378be3182f916709e8d06bd955232e650a7188b86","side":"right"},{"sibling":"d92781c59301ffd5bfb0bad75d9fdf6d73879f715518149d383f7362213e0daf","side":"right"},{"sibling":"f315303d4402b57497416d48eb4c4bb50405b40862d41c7caf318cb3d29c5237","side":"right"},{"sibling":"36973eb5f586cd67e0c0dc055dd87e734aa544c35d4400d2c9f932f8ef8fb27f","side":"right"},{"sibling":"e39f7900355489c4718b21cc2d3d06382e1d2f22b864d10be9ebc9d498b279a1","side":"right"},{"sibling":"1f9a970b25dd938c98cabc9e0a55c5a6f46292b9fa37cc89d90ef0cbb1e05a8c","side":"left"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"3b50864499c874394ea0928567747666eaf59b01380e46cd52164ec5acec0f71","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":361008,"merkle_root":"3065e8369ea437c06beba806dc4e4bb159979adeb21fe632242c1906a7204647","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260728T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-28T05:38:48Z","sig_algorithm":"ed25519","signature":"9141644407577a82611c1579110f667de2d46dc6b93c6322edf26f4c3056ea99f0e56502853908e30d87c38bcf95eb6e0ab5130525aa51505bd6f61938120609","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_7d1d8b94ceef385b6ed62b169726d78400a583bf904d4ab089bdcee09e27ccbf"}}