{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_e0fa8a6ed423c9bccb65503b19c6160e48e065be1ec0339799aa06cfff394a99","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_e0fa8a6ed423c9bccb65503b19c6160e48e065be1ec0339799aa06cfff394a99","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"dd08860e158d1d1e814fb38d4428bab68aa05ca38a0e269a14c9e0a90f284b2e","published":"Tue, 26 May 2026 00:00:00 -0400","receipt_hash":"dd08860e158d1d1e814fb38d4428bab68aa05ca38a0e269a14c9e0a90f284b2e","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"dd08860e158d1d1e814fb38d4428bab68aa05ca38a0e269a14c9e0a90f284b2e","observed_at":"2026-05-26T04:43:39.018238Z","parent_run_hash":"dca8dedd754ad6a1772113d6b97ee4f4ab9a0afbdeb44aaace5ff2d2446b164b","published":"Tue, 26 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.25850v1 Announce Type: cross \nAbstract: This paper investigates large language model (LLM) abstention learning, specifically using ternary reward, which incentivize truthfulness in large language models. This paper extends that idea by moving from a ternary reward to a Trajectory-Informed advantage reweighting, dynamically re-weights the abstention reward during Group Relative Policy Optimization (GRPO) training. The objective of this work focuses on abstention learning instead of improving truthfulness, serving as an exploration into hallucination reduction. The novelty of this paper lies in methodological innovation, advantage re-weighting, and benchmark selection. Leveraging GRPO's multiple trajectories as a natural abstention signal, this method uses a reward signal to explore knowledge boundaries and encourage consistency. By demonstrating that trajectories can be used as a confidence indicator of the policy relative to the query, they are then used to dynamically calcu","title":"TIAR: Trajectory-Informed Advantage Reweighting for LLM Abstention Learning","url":"https://arxiv.org/abs/2605.25850","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.25850v1 Announce Type: cross \nAbstract: This paper investigates large language model (LLM) abstention learning, specifically using ternary reward, which incentivize truthfulness in large language models. This paper extends that idea by moving from a ternary reward to a Trajectory-Informed advantage reweighting, dynamically re-weights the abstention reward during Group Relative Policy Optimization (GRPO) training. The objective of this work focuses on abstention learning instead of improving truthfulness, serving as an exploration into hallucination reduction. The novelty of this paper lies in methodological innovation, advantage re-weighting, and benchmark selection. Leveraging GRPO's multiple trajectories as a natural abstention signal, this method uses a reward signal to explore knowledge boundaries and encourage consistency. By demonstrating that trajectories can be used as a confidence indicator of the policy relative to the query, they are then used to dynamically calcu","title":"TIAR: Trajectory-Informed Advantage Reweighting for LLM Abstention Learning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-26T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.25850"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:947224f728009a6e46194faf58c458c25b01e7bebe726a1973274a1fe6634e9dc506d54093707e423a598a57711782be056322ac058d6d77c02bfdbf1aed3508","signer":"crovia.substrate","subject":{"observed_at":"2026-05-26T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.25850"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f72f2c1cbd60dd3ed8b832ac8ebad765c4bcd90f3a9cf5420a40a740fe305cf1","leaf_index":151736,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"869808c4b0d412ed0c7e3625bca6a6b098dca3b617b9960f5a372de79e0a93e1","side":"right"},{"sibling":"85506dfbe059d6631970cdd719c2c9f8ea8d0ab8665590ffbcdbb5573166b689","side":"right"},{"sibling":"20ccc7ff1f03a2d60fa1b8728d4c19f8cdc9d698c1cfe6295c0c16bdba285564","side":"right"},{"sibling":"3b0d59885a736905c4146a45399a8b6c26ed07416340c1b9b8c69ff33845cfd8","side":"left"},{"sibling":"a5902b876622bd211b3c9e464511b2e35b3065be489c1576ee6f299c1855d2f1","side":"left"},{"sibling":"1fdb62cc53d97a7f7a1438e1d514a795206e3fd636e208e00b18fba2519cd611","side":"left"},{"sibling":"59c025ecd2dec1777c83a364f63a0af602c1b8c1069e8b09513a2e68a6f64e32","side":"right"},{"sibling":"f9db7ff3c5db7011b2738b446bad7b165ec8fd162c800ba3bad08298201edaa3","side":"left"},{"sibling":"aff657821100efc777fe98abc63d25a13c2d813d2a00f85a278e295ee3a166b3","side":"right"},{"sibling":"879666fab72e779ab55d0564eaabd64b00534fd6bba7f18412c7f31f61ffd09f","side":"right"},{"sibling":"f40ccedd90c323817e961adc0a2e2db82b8aabe192b6c9d5a373ff988987b207","side":"right"},{"sibling":"b85ea61ae405eed84392a7b6b1eee5536f5a38d6b04070638e23ec7b71e3443a","side":"right"},{"sibling":"d415e6939aee710631f5062799379b547d2c3e3d9a68f263bbb5a693285ab2ca","side":"left"},{"sibling":"e5893793e3591ed7f5e58ca94ffcfba46bb30f69fb1c25d5ba8ef49eb99f9126","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"e3eecaf996dbe7229a7bb1d234c97aea97a252c5f7c89f8547b6d091db0f0e40","side":"right"},{"sibling":"55bcbd4da3e20d93931f7e58673f10232e81a5b1514d7396cb4b71e8f95788d0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":152106,"merkle_root":"ac5182c6f3dd09931f2a689df5f4be36df7b55e55bcf106e195671f5ed55fd8f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260526T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-26T05:37:34Z","sig_algorithm":"ed25519","signature":"a401243fbd2c077d29a623c6ef616c78fbd5c8ce1930afa7af7165b2386e5a8cf15d5094983a1c962e71b27b911a3f3ce09c8cfab9393be2ce5ce8ee6513da06","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_e0fa8a6ed423c9bccb65503b19c6160e48e065be1ec0339799aa06cfff394a99"}}