{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_441e317a42ea35a8c8799fa5d19bd61de2138e85c7f40c22f091d9a0a0e326d7","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_441e317a42ea35a8c8799fa5d19bd61de2138e85c7f40c22f091d9a0a0e326d7","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"938060bb36d01616dc68881a50f95df6887a0db9a0bb242e6afcf9670027ee56","published":"Fri, 22 May 2026 00:00:00 -0400","receipt_hash":"938060bb36d01616dc68881a50f95df6887a0db9a0bb242e6afcf9670027ee56","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"938060bb36d01616dc68881a50f95df6887a0db9a0bb242e6afcf9670027ee56","observed_at":"2026-05-22T04:43:12.593252Z","parent_run_hash":"dd4d56660d55b2d65dc84dc5e7c8f83487d90da2dd343b5707d6948a3bb0d917","published":"Fri, 22 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.20256v1 Announce Type: cross \nAbstract: Reinforcement learning has become a cornerstone for aligning and unlocking the reasoning capabilities of large-scale models. At its core, the training loop of GRPO and its variants alternates between rollout sampling and policy update. Unlike supervised learning, where each gradient step is anchored to an explicit ground-truth target, the optimal gradient direction for updating model parameters in this setting is not known a priori; the high-quality rollouts drawn during the sampling stage therefore act as the implicit \"teacher\" that guides every parameter update. However, GRPO adopt a simple sampling scheme that conditions all rollouts on the same original prompt. When a task lies beyond the policy model's current capability, this sampling scheme rarely yields a high-quality rollout, leaving the policy model without a meaningful gradient direction when updating its parameters, which causes training to stall. To address this issue, we ","title":"FBOS-RL: Feedback-Driven Bi-Objective Synergistic Reinforcement Learning","url":"https://arxiv.org/abs/2605.20256","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.20256v1 Announce Type: cross \nAbstract: Reinforcement learning has become a cornerstone for aligning and unlocking the reasoning capabilities of large-scale models. At its core, the training loop of GRPO and its variants alternates between rollout sampling and policy update. Unlike supervised learning, where each gradient step is anchored to an explicit ground-truth target, the optimal gradient direction for updating model parameters in this setting is not known a priori; the high-quality rollouts drawn during the sampling stage therefore act as the implicit \"teacher\" that guides every parameter update. However, GRPO adopt a simple sampling scheme that conditions all rollouts on the same original prompt. When a task lies beyond the policy model's current capability, this sampling scheme rarely yields a high-quality rollout, leaving the policy model without a meaningful gradient direction when updating its parameters, which causes training to stall. To address this issue, we ","title":"FBOS-RL: Feedback-Driven Bi-Objective Synergistic Reinforcement Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-22T04:43:12Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.20256"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b537e17a75ce2913e2451abc42f343c8379205bdb4d8cb23bf3fb0c777bd9fea9829edcb001b3a1412f40d636140388ef4553f4709443a420199338e01655e01","signer":"crovia.substrate","subject":{"observed_at":"2026-05-22T04:43:12Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.20256"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"aa448d7d3e4482bf8cbfd44ec6f800155dc073234add05cc4c9fdec773320a11","leaf_index":148180,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"65cd02a73e4efe7f6d55bd987ba6182062fddd7b12af2103c724ce285d145c7d","side":"right"},{"sibling":"a6dfc9a5c9bfdfb2fd2c303a53bf26066486c6eb8a628f5d4780ab44aba32447","side":"right"},{"sibling":"5959ab1e02b830d168925a9301779d7921db08e1b05803c1c5bc9ab1ab8c67f6","side":"left"},{"sibling":"77466f444487fbe1510677b2c2d99cb388d4594ccf5ecd926031232d828bd4c9","side":"right"},{"sibling":"e515ecd738733f891313e29360912c2fd7b26e7dda403946d6d8258b563289fd","side":"left"},{"sibling":"0fdd2a5a527954fdd3c361b2337a8a7abc44029bc78402b0f1c47428a708f845","side":"right"},{"sibling":"9041a5a271687216ad8ad42ade0fd8c711a35ed68eb350b81aed4384188eec3c","side":"left"},{"sibling":"ce8423e7b33fd98ad2188ec515d860afebe3ce0e74717c41376c90ea6acfb384","side":"left"},{"sibling":"c5d582bc1cdd6d6494fd9e29c8b4aadd02d777f7ef92fc6d4afad7ee42785e79","side":"right"},{"sibling":"d337a9fdcfc121e9691d8db9173af6a3fe0c33d4a6d5f0c8a7a01af04f9fb856","side":"left"},{"sibling":"8b39e07457f5cc5d687d2ae42284dbe705bb87084e7db4b626aff81e51dacd19","side":"right"},{"sibling":"79a713e1e345ccb99c5fe994a11708c8e9bcfa2e91f940d70421cb7d8d77ecc6","side":"right"},{"sibling":"249870fb494bef050c409081e5de45f9042938d7b2823524ee496f296aa63667","side":"right"},{"sibling":"96c48ee8328f1b7925a4cc4421df5cb0bd81c92a5d8354c93126fa5f0166d225","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"4e13b4a3e69bb83d13913477a782913ce03937edd65046853c1964d3cbb6564b","side":"right"},{"sibling":"0f7b2df1c4580bf7bb7c24b9158ba20593a06af18a5f18d0973e5eff20c35cd8","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":148601,"merkle_root":"44900cffd986f40535c83f46a250e86fbf1019d41f27080a00fbf9b8d77ec33a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260524T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-24T13:37:32Z","sig_algorithm":"ed25519","signature":"059e428c3c5241de303721ad6ac7b748758372180f3f0a717810312aecd6fab073abb3ea264157666de581a2b361c4c6e2aa8ca081b08ce9be090721f0e3400e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_441e317a42ea35a8c8799fa5d19bd61de2138e85c7f40c22f091d9a0a0e326d7"}}