{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_fb8fd0b1f03e7e930fd628c0cb89a9bd853db616d1aff99a6a4a7f16229a5e67","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_fb8fd0b1f03e7e930fd628c0cb89a9bd853db616d1aff99a6a4a7f16229a5e67","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"817dc19507e8bd6eabd240d7c2ab4c3000f6733d6410cbd33bea7ff90df8acc7","published":"Fri, 12 Jun 2026 00:00:00 -0400","receipt_hash":"817dc19507e8bd6eabd240d7c2ab4c3000f6733d6410cbd33bea7ff90df8acc7","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"817dc19507e8bd6eabd240d7c2ab4c3000f6733d6410cbd33bea7ff90df8acc7","observed_at":"2026-06-12T04:43:44.933383Z","parent_run_hash":"a8b304a31db3809a528f5a45e58597f7bb9f23028e53b4f9ed2f5599dd731b5e","published":"Fri, 12 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.12479v1 Announce Type: cross \nAbstract: Large language model (LLM) routing has emerged as an effective paradigm for leveraging the complementary strengths of multiple LLMs through dynamic model and reasoning-strategy selection. Recent reinforcement learning (RL)-based routing methods further improve routing quality by optimizing routing policies from interaction feedback. However, they still struggle to provide informative and comparable learning signals under heterogeneous tasks with varying difficulty. In practice, multiple objectives (e.g., correctness, format behavior) are aggregated into a single scalar reward, leading to ambiguous credit assignment and conflicting optimization signals. Moreover, reward signals exhibit significant variability across instances, where some instances produce higher or more variable rewards, introducing optimization bias that favors trivial samples over informative ones. To address these issues, we propose \\textbf{ReCal}, a \\textbf{\\underli","title":"ReCal: Reward Calibration for RL-based LLM Routing","url":"https://arxiv.org/abs/2606.12479","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.12479v1 Announce Type: cross \nAbstract: Large language model (LLM) routing has emerged as an effective paradigm for leveraging the complementary strengths of multiple LLMs through dynamic model and reasoning-strategy selection. Recent reinforcement learning (RL)-based routing methods further improve routing quality by optimizing routing policies from interaction feedback. However, they still struggle to provide informative and comparable learning signals under heterogeneous tasks with varying difficulty. In practice, multiple objectives (e.g., correctness, format behavior) are aggregated into a single scalar reward, leading to ambiguous credit assignment and conflicting optimization signals. Moreover, reward signals exhibit significant variability across instances, where some instances produce higher or more variable rewards, introducing optimization bias that favors trivial samples over informative ones. To address these issues, we propose \\textbf{ReCal}, a \\textbf{\\underli","title":"ReCal: Reward Calibration for RL-based LLM Routing","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-12T04:43:44Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.12479"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1e30c9456fc97bf21781263f13fe3f2d8218edf88b8aa0728dfc4c6ff60b8c07e4ccfdb5ebb1a5b4edd8ec537160d42807a298547ac4ea5e51618cfb7ec8c40d","signer":"crovia.substrate","subject":{"observed_at":"2026-06-12T04:43:44Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.12479"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7ac5e26cff03fc7ca977600901165195fc1525a3a06fc670b0690701cef21f1b","leaf_index":229690,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"27d605260f754513ca54495b664744d3efc21e5563445fb6d51e6f0c4482404a","side":"right"},{"sibling":"d404962138215a338a96b9cc5f5de515641e6d06d9967244266ac0c08a13b3ef","side":"left"},{"sibling":"7f0f6f4b556b1e204164ee33ab7f8dc63f9ee72db343730fcd47efa4f23f82ec","side":"right"},{"sibling":"0548e0ea5426f2cdc2573bd7d124aea04584ada8be6c3b230f67a1dd2f999e0c","side":"left"},{"sibling":"2e90a78e0915fe5d6039975f5cb01a939386d4500959fe89c819c6cbf7d9e037","side":"left"},{"sibling":"55f2cc8a9488d1be607c481b52c6c043cb62aedd1173f4cdb41b1fbcf8a2574d","side":"left"},{"sibling":"a6a078de209021e47c44d3d118668e1ccb7ae97cf5ec348f1b84c3f8cf25571a","side":"right"},{"sibling":"2654c48172d365d1cff889fbe9c8f8e481d6ccd161a53a1f187cd0415e9830c1","side":"right"},{"sibling":"99d288e6cd43a807fba958865176bb1c08b82471afe942f6e4989eaeeb7275aa","side":"left"},{"sibling":"d385017d38a86f6abc492026a7cd60ceb3b3ff2486142dc49e5acb179f4b7d11","side":"right"},{"sibling":"bde25d7e94e64717e426a97f6fcb4907e92b5c61fc89d92d7e0947a2249c3f6b","side":"right"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_fb8fd0b1f03e7e930fd628c0cb89a9bd853db616d1aff99a6a4a7f16229a5e67"}}