{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_e4e3b514689c6c356e2e8c5c4f78fa78713906bb8aa2211e4b7f3f78d893807d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_e4e3b514689c6c356e2e8c5c4f78fa78713906bb8aa2211e4b7f3f78d893807d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"176666886c1d619737e26d0517f446cab5fb0fd3f498d62a292022ab3872b142","published":"Fri, 15 May 2026 00:00:00 -0400","receipt_hash":"176666886c1d619737e26d0517f446cab5fb0fd3f498d62a292022ab3872b142","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"176666886c1d619737e26d0517f446cab5fb0fd3f498d62a292022ab3872b142","observed_at":"2026-05-15T04:43:17.611638Z","parent_run_hash":"5c64f85625fabd323e9c4a1cf068c012fb88a248deda9a9ac702fb2f9799f2e5","published":"Fri, 15 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.04265v3 Announce Type: replace-cross \nAbstract: Reinforcement Learning with Verifiable Rewards (RLVR) has emerged as a promising paradigm for enhancing reasoning in Large Language Models (LLMs). However, existing reward formulations typically treat exploration and consolidation as a monolithic process, resulting in entangled stage-wise learning dynamics. This contradicts the natural learning behavior of human learners. In human learning, individuals adopt distinct behavioral patterns toward mastered versus unfamiliar problems. When confronting unmastered challenges, humans prioritize broad exploration to seek viable solutions. By contrast, for well-mastered problems, they focus instead on reasoning condensation and knowledge abstraction to distill concise underlying principles. Motivated by this gap, we introduce T2T(Thickening-to-Thinning), a dynamic reward framework inspired by human learning processes. Specifically, it implements a dual-phase mechanism: (1) On incorrect a","title":"Boosting LLM Reasoning via Human-Inspired Reward Shaping","url":"https://arxiv.org/abs/2602.04265","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.04265v3 Announce Type: replace-cross \nAbstract: Reinforcement Learning with Verifiable Rewards (RLVR) has emerged as a promising paradigm for enhancing reasoning in Large Language Models (LLMs). However, existing reward formulations typically treat exploration and consolidation as a monolithic process, resulting in entangled stage-wise learning dynamics. This contradicts the natural learning behavior of human learners. In human learning, individuals adopt distinct behavioral patterns toward mastered versus unfamiliar problems. When confronting unmastered challenges, humans prioritize broad exploration to seek viable solutions. By contrast, for well-mastered problems, they focus instead on reasoning condensation and knowledge abstraction to distill concise underlying principles. Motivated by this gap, we introduce T2T(Thickening-to-Thinning), a dynamic reward framework inspired by human learning processes. Specifically, it implements a dual-phase mechanism: (1) On incorrect a","title":"Boosting LLM Reasoning via Human-Inspired Reward Shaping","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-15T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.04265"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:58a92d311df0a79c6e7282c6b2847339196c746938a0d332b05a0ec26f13451b3cc1c4b6ac65c73b3b3d147facad28ce4af2ff7897cbf6cc05ec96d3223b3b07","signer":"crovia.substrate","subject":{"observed_at":"2026-05-15T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.04265"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"cdea9b924275d2e2aefc35f107f4d76938f3fea9d25ae85dc76d6755e7e5dcdc","leaf_index":134764,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f78aad56f24f8ed1a1ed35bac97eb43944a4befab91cf4c3a6119164007affd8","side":"right"},{"sibling":"8b4cc5a606ad9007a9dcf23a285890ced3ac26703a8681b9fa08b524d2744e11","side":"right"},{"sibling":"a227dd46da69a4487a52dbfbda6a5e95a93ca4a3b3105d13ea31f6446d6b342d","side":"left"},{"sibling":"f8d54f25c5c6a3899698f1b719457c48a0191fc6d08ac6c40f72e617359b8ffd","side":"left"},{"sibling":"37ba49eedf02189e9205dda7034b06c5dea5c0b014937cfd3d4b16e4d6725e50","side":"right"},{"sibling":"952bb2ea9b8e1bb4f01c53d0e4d6522140e49683ac624344442d5a1522d7c3be","side":"left"},{"sibling":"4999a953ceed25471a2bccd6eadce4cee20773995bf2d506b76320a8705da9a4","side":"left"},{"sibling":"d8ac22bfe982d5dc7f00879b0b1dfae8f4e95a38a6242933cbebb3b7271d5704","side":"right"},{"sibling":"a17a0eaa87574a308e2c02cf125072b9e69566b9e8d6d2e64a02880192b885d5","side":"right"},{"sibling":"6bd0475fd3a73322a4f73095c96b072de89e462ce5839161aa552e0208619bce","side":"left"},{"sibling":"36672459e5ed50c64ee1842b69cb6d2eb682c2a04844555be8d124257571994a","side":"left"},{"sibling":"727783827652adfa99c455bd80a01bfb33836228e51068b4f654ef3da468ca69","side":"left"},{"sibling":"623194cd30880ed223e306737fdb111aa0d781751bfc47553c404a6af6aad2c4","side":"right"},{"sibling":"fc8f53ed42756907fb79ee19a4ed09f72c560e5302b3d98198b96bf1da635a4a","side":"right"},{"sibling":"d6607539da7ba39ec68be2d12f27ed6768766c745e3120fd915f88c5e288e07c","side":"right"},{"sibling":"b63408a424d27cd6a75e0fb155e69a58a328f41e9cb9dba1eddef9a5289cc7fd","side":"right"},{"sibling":"356fb36a4e188f03d7a05c54cd8789bdd40eda454b9bc9560f667acc08e6c4e0","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":134886,"merkle_root":"c6c7ae28c065bced89e7f844216b073f1a7cc4b378db0d41a98bcd21b28066db","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260515T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-15T05:37:26Z","sig_algorithm":"ed25519","signature":"5a3978c26017daf4104adbb3e1c3099c5750157acbfc7242ece1815dc6740fe08a690291ce0afe42011e20cc565b5ebe64ec016bf658b5bab563a63337985c05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_e4e3b514689c6c356e2e8c5c4f78fa78713906bb8aa2211e4b7f3f78d893807d"}}