{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_93cc563f587f427af7acb62a237ad8f0c315d8134c2f04a860edb6d9a4e7e50e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_93cc563f587f427af7acb62a237ad8f0c315d8134c2f04a860edb6d9a4e7e50e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"8be8561fc1fa58caa1be0654703eff4483ca841d40ebd42fce9ba3c103685c38","published":"Mon, 01 Jun 2026 00:00:00 -0400","receipt_hash":"8be8561fc1fa58caa1be0654703eff4483ca841d40ebd42fce9ba3c103685c38","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"8be8561fc1fa58caa1be0654703eff4483ca841d40ebd42fce9ba3c103685c38","observed_at":"2026-06-01T04:43:13.859018Z","parent_run_hash":"8993bbc535dae8c9669e099af3624cb39166b8d9bbfd66f26ae5c338cbb21be2","published":"Mon, 01 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.30903v1 Announce Type: cross \nAbstract: Inverse reinforcement learning (IRL) typically assumes demonstrations from a single optimal demonstrator, but in many applications data come from multiple imperfect demonstrators with heterogeneous suboptimality levels. We study reward learning in this setting through a feasible-reward-set framework: for each demonstrator, we encode its declared suboptimality level as a linear constraint and intersect the resulting feasible sets across demonstrators. Our theoretical analysis shows that the joint feasible set shrinks monotonically as data are added, and we give an exact characterization of when a new demonstrator strictly tightens it. We further establish two recovery guarantees for the feasible reward set of the ground-truth optimal demonstrator: one bound depends on closeness to the optimal occupancy, while the other requires only sufficient coverage and no near-optimal demonstrator. On the practical side, we introduce strategies to a","title":"Inverse Reinforcement Learning without an Optimal Demonstrator: A Feasible Reward Set Approach","url":"https://arxiv.org/abs/2605.30903","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.30903v1 Announce Type: cross \nAbstract: Inverse reinforcement learning (IRL) typically assumes demonstrations from a single optimal demonstrator, but in many applications data come from multiple imperfect demonstrators with heterogeneous suboptimality levels. We study reward learning in this setting through a feasible-reward-set framework: for each demonstrator, we encode its declared suboptimality level as a linear constraint and intersect the resulting feasible sets across demonstrators. Our theoretical analysis shows that the joint feasible set shrinks monotonically as data are added, and we give an exact characterization of when a new demonstrator strictly tightens it. We further establish two recovery guarantees for the feasible reward set of the ground-truth optimal demonstrator: one bound depends on closeness to the optimal occupancy, while the other requires only sufficient coverage and no near-optimal demonstrator. On the practical side, we introduce strategies to a","title":"Inverse Reinforcement Learning without an Optimal Demonstrator: A Feasible Reward Set Approach","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-01T04:43:13Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.30903"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:8d58e9334d5ec6607bb3491d8148dd894de5b91c767b8ee212cf6db37890b837b625a533345b82b12ce85284c5b173d480a339de240cbf42ca0ea200e4c00804","signer":"crovia.substrate","subject":{"observed_at":"2026-06-01T04:43:13Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.30903"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"4b20f28aeab177ce3755e8280e41328c2acadb0dfba7ecff4a08371796595687","leaf_index":163890,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"69efaa556755062140077922314030de9ba907c1bb67913fee47c352a910d043","side":"right"},{"sibling":"4a372b84532f79ec9f4c578ad037b86092176ea640605e72be97caa8324c1d41","side":"left"},{"sibling":"0e3ef5b2e701e28956860d6f130f5ebaa9bec29a93570c87b56515a0fded423b","side":"right"},{"sibling":"09e63001d80d81a478b37369e477777c6b0a4fdc26a342cf7638f514600e6dd6","side":"right"},{"sibling":"5e0ac2c9324b343ecacb65a874281d7d5e47f2eb5e8ad665d94fcd3133fac7e0","side":"left"},{"sibling":"6d2f0e650ee7cbadfa1d8b537c7ba55f2431da2ab6c87c375b51a57393551957","side":"left"},{"sibling":"8759b29a2f5a0cbc7e311e83a6c64093a22568f86085e1c65bead39ebc241c88","side":"right"},{"sibling":"c80f3e378be31136ba3005ae1d4e6fb73ee42b46d44ded3d5ec7b012545b00f5","side":"right"},{"sibling":"86fede29e507d01bfe35484a1579149e8e99b3527ff48559423f6056d11af46a","side":"right"},{"sibling":"fc601f0745c37fc5f7c249300e654db05d61bb9059885eeb0b0a5047b5d28408","side":"right"},{"sibling":"6ed9290cdae063f13bfeb71c4c5440595cabb61fd4b39225b9cc913ffb336dd7","side":"right"},{"sibling":"a0446b923d1ce90021e78edff07f6bfc7cc2a1326a565c5f0786b82a24dd0a2a","side":"right"},{"sibling":"e598fd53912c30e58ca8e58d7d8a338fe0f2ecb63fdd99225bc703c499c948ec","side":"right"},{"sibling":"fc4873333221ec8167697f75b6f6a8a08491a8cf18952defb65fb6d4958fa5e7","side":"right"},{"sibling":"05c8a827da2a05549ee3250310777009120c687885816bf6c7c74801bfaa346d","side":"right"},{"sibling":"5ea2f2dc9f046df723b6bd9932d61a9a3d80a76e79ce1b939b3d93ff79b5a91c","side":"left"},{"sibling":"ce41d9b82f34b16efd653dfb3552acc4e2512939e47903e5fc979fbed00c5764","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":164217,"merkle_root":"a1098816aea1b60b8fe37b62410469bc5024a2c335bbec4f6ef2add7875dbdf2","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260601T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-01T05:37:41Z","sig_algorithm":"ed25519","signature":"d7f91db1d54b9495c499440c2828f4bd53360555391ce6e25adea5183bc1fa0f697d80708a099d0b0429e6f8cb6c71e7fccf82acb3c84481149974fb26074708","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_93cc563f587f427af7acb62a237ad8f0c315d8134c2f04a860edb6d9a4e7e50e"}}