{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_3533f30f516dbb6255de63e08a686d8ea32453b071c19028c5415e47cc690c40","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_3533f30f516dbb6255de63e08a686d8ea32453b071c19028c5415e47cc690c40","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"1d8809dd67be1a17dca55af1ed373f2026c72f27b64538f8d2027a9c2577854d","published":"Mon, 01 Jun 2026 00:00:00 -0400","receipt_hash":"1d8809dd67be1a17dca55af1ed373f2026c72f27b64538f8d2027a9c2577854d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"1d8809dd67be1a17dca55af1ed373f2026c72f27b64538f8d2027a9c2577854d","observed_at":"2026-06-01T04:43:13.859018Z","parent_run_hash":"8993bbc535dae8c9669e099af3624cb39166b8d9bbfd66f26ae5c338cbb21be2","published":"Mon, 01 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2511.04393v2 Announce Type: replace \nAbstract: Large language models (LLMs) are increasingly deployed as \"agents\" for decision-making (DM) in interactive and dynamic environments. Yet, since they were not originally designed for DM, recent studies show that LLMs can struggle even in basic online DM problems, failing to achieve low regret or an effective exploration-exploitation tradeoff. To address this, we introduce Iterative Regret-Minimization Fine-Tuning (Iterative RMFT), a post-training procedure that repeatedly distills low-regret decision trajectories back into the base model. At each iteration, the model rolls out multiple decision trajectories, selects the k-lowest regret ones, and fine-tunes itself on them. Unlike prior methods that (a) distill action sequences from known DM algorithms or (b) rely on manually crafted chain-of-thought templates, our approach leverages the regret metric to elicit the model's own DM ability and reasoning rationales. This reliance on model-","title":"Post-Training LLMs as Better Decision-Making Agents: A Regret-Minimization Approach","url":"https://arxiv.org/abs/2511.04393","vendor":"arxiv_cs_ai"},"summary":"arXiv:2511.04393v2 Announce Type: replace \nAbstract: Large language models (LLMs) are increasingly deployed as \"agents\" for decision-making (DM) in interactive and dynamic environments. Yet, since they were not originally designed for DM, recent studies show that LLMs can struggle even in basic online DM problems, failing to achieve low regret or an effective exploration-exploitation tradeoff. To address this, we introduce Iterative Regret-Minimization Fine-Tuning (Iterative RMFT), a post-training procedure that repeatedly distills low-regret decision trajectories back into the base model. At each iteration, the model rolls out multiple decision trajectories, selects the k-lowest regret ones, and fine-tunes itself on them. Unlike prior methods that (a) distill action sequences from known DM algorithms or (b) rely on manually crafted chain-of-thought templates, our approach leverages the regret metric to elicit the model's own DM ability and reasoning rationales. This reliance on model-","title":"Post-Training LLMs as Better Decision-Making Agents: A Regret-Minimization Approach","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-01T04:43:13Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2511.04393"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:84d0188f4002ebf1bffff3b79f30854565b03672612b74ed3d71fd94879571a73acd54bd1f254415cb2f6b6b87bc8805df9032549d4ddb6c6a4a05bf4751d106","signer":"crovia.substrate","subject":{"observed_at":"2026-06-01T04:43:13Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2511.04393"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8c7aa6d9a4ae0e3dae7178253629a56464dfcda87325b7e23d87c7a354666a9e","leaf_index":164001,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"6764c812a66b3016cadaabcf543162d43baf08f0e87d132f0c73a3c77c710500","side":"left"},{"sibling":"d4e7dd54821fb01b9961bca3662c73cd4430c37c144c24d8bf548c8ded559ab2","side":"right"},{"sibling":"81d8f032b848ae730b7e76c698928dc80515f559458c2a056eb9f98fd5aab371","side":"right"},{"sibling":"55f15a8c1578a5293965661a23dd862f90d71b9e98e657628a684c66a7a5b7ed","side":"right"},{"sibling":"2f465fb071a98a6ea8b6b765aa4b8bf2624470f1f846713514a21bfd02d575e5","side":"right"},{"sibling":"2ed5dc81b48343dc042200ddd691ec7502014cec7a175df1a1ffd9a76812eb3a","side":"left"},{"sibling":"1d23b94e7765a5d6bcc243f4f7ecca5895f450b29cd855b417c9c7111e6b5a45","side":"right"},{"sibling":"c2ed258674acbec09c8c9dbf15c89e6fc50854c6eb5f8d42b378de7db248641a","side":"left"},{"sibling":"86fede29e507d01bfe35484a1579149e8e99b3527ff48559423f6056d11af46a","side":"right"},{"sibling":"fc601f0745c37fc5f7c249300e654db05d61bb9059885eeb0b0a5047b5d28408","side":"right"},{"sibling":"6ed9290cdae063f13bfeb71c4c5440595cabb61fd4b39225b9cc913ffb336dd7","side":"right"},{"sibling":"a0446b923d1ce90021e78edff07f6bfc7cc2a1326a565c5f0786b82a24dd0a2a","side":"right"},{"sibling":"e598fd53912c30e58ca8e58d7d8a338fe0f2ecb63fdd99225bc703c499c948ec","side":"right"},{"sibling":"fc4873333221ec8167697f75b6f6a8a08491a8cf18952defb65fb6d4958fa5e7","side":"right"},{"sibling":"05c8a827da2a05549ee3250310777009120c687885816bf6c7c74801bfaa346d","side":"right"},{"sibling":"5ea2f2dc9f046df723b6bd9932d61a9a3d80a76e79ce1b939b3d93ff79b5a91c","side":"left"},{"sibling":"ce41d9b82f34b16efd653dfb3552acc4e2512939e47903e5fc979fbed00c5764","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":164217,"merkle_root":"a1098816aea1b60b8fe37b62410469bc5024a2c335bbec4f6ef2add7875dbdf2","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260601T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-01T05:37:41Z","sig_algorithm":"ed25519","signature":"d7f91db1d54b9495c499440c2828f4bd53360555391ce6e25adea5183bc1fa0f697d80708a099d0b0429e6f8cb6c71e7fccf82acb3c84481149974fb26074708","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_3533f30f516dbb6255de63e08a686d8ea32453b071c19028c5415e47cc690c40"}}