{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_18b97dbe6f10419fb49f0c229eb52a852a0b0c67887944bd91495022de1e6f30","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_18b97dbe6f10419fb49f0c229eb52a852a0b0c67887944bd91495022de1e6f30","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"20b34037817d771b903821a429ff7d375d2ed26143622e979ca2fd0b523ff79d","published":"Wed, 24 Jun 2026 00:00:00 -0400","receipt_hash":"20b34037817d771b903821a429ff7d375d2ed26143622e979ca2fd0b523ff79d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"20b34037817d771b903821a429ff7d375d2ed26143622e979ca2fd0b523ff79d","observed_at":"2026-06-24T04:43:17.877668Z","parent_run_hash":"ca17d06d44ba7db934e6f913874699efc608b8f87453f1ac67f52060e620b57c","published":"Wed, 24 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.24622v1 Announce Type: new \nAbstract: Training safe Reinforcement Learning (RL) systems is inherently challenging, with no guarantee of avoiding unwanted behaviors. The most effective defenses against this are (i) transparency through explainability and (ii) alignment via human feedback. While both show promising results, no publicly available framework currently combines them. To address this, we introduce Themis, an XAI-enabled testing and evaluation framework for Reinforcement Learning from Human Feedback. Themis supports over 200 widely used environments and is easily configurable for experiments in RL, transparency, and alignment. Our results show that Themis can train reward models that match or outperform the environment's true reward signal using human preferences. We also provide a cloud-based platform for collecting human feedback and managing experiments. It is user-friendly, auto-scalable, and supports large participant groups across multiple experiments without ","title":"Themis: An explainable AI-enabled framework for Reinforcement Learning with Human Feedback","url":"https://arxiv.org/abs/2606.24622","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.24622v1 Announce Type: new \nAbstract: Training safe Reinforcement Learning (RL) systems is inherently challenging, with no guarantee of avoiding unwanted behaviors. The most effective defenses against this are (i) transparency through explainability and (ii) alignment via human feedback. While both show promising results, no publicly available framework currently combines them. To address this, we introduce Themis, an XAI-enabled testing and evaluation framework for Reinforcement Learning from Human Feedback. Themis supports over 200 widely used environments and is easily configurable for experiments in RL, transparency, and alignment. Our results show that Themis can train reward models that match or outperform the environment's true reward signal using human preferences. We also provide a cloud-based platform for collecting human feedback and managing experiments. It is user-friendly, auto-scalable, and supports large participant groups across multiple experiments without ","title":"Themis: An explainable AI-enabled framework for Reinforcement Learning with Human Feedback","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-24T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.24622"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:e9e4a91eb606c2a0d69db4442d6046884bd1364dd99d6d3098455e35715eedd2373e94b228cecee006f50c0b7892cf747d3051f352ff5343b380a587901a7b0a","signer":"crovia.substrate","subject":{"observed_at":"2026-06-24T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.24622"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8f7348774d198b61761b00309724fdaa64abe09fab58926d6d9ff1693696ac97","leaf_index":244466,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"2766a6caff77245bc99a2d1446527512e18554a274111e49db6177b19db28297","side":"right"},{"sibling":"f379fd619f57e172684740d58f700ad1d21d6e6cb681cfb3a33958e13c7f15ed","side":"left"},{"sibling":"d654892a24b32d867b9a849b5525c543372dbeb84ebdee6a9a7a9bb155b092b5","side":"right"},{"sibling":"52d8694ec4a522884d52ac6a66373ac233c68a747a6de4becad700a70c70e95b","side":"right"},{"sibling":"43402a0314012098dc5f6d78feb0aa1deac3dfa8f9f12d5ffbcbb9e2b02f51d0","side":"left"},{"sibling":"6b79a26c3c663a3456bb33f67f4eb110b54a7864e536636d6d6a5f1ecb5b4dac","side":"left"},{"sibling":"c56732dc17eeb36165904e6852de6bba2b9a84b8469f41c4ec2f52d30d7ec552","side":"left"},{"sibling":"15f48e0a2913d128f60a5e7f790a33afe2bc5de6c7d7d3be71d0b93c1434d471","side":"left"},{"sibling":"2b242921d744c9b6cbe2a988917698dfd00620e46afc5362a806f02ffdd1f721","side":"right"},{"sibling":"bc74ebb08462da8a50fc65ea75f8a8a3418d10ebd471d830f1c67f33dd54dfd1","side":"left"},{"sibling":"6dafd355e5d54c60e61c6c02d3842984e234b1f5bca1623fcdd3def7b8931973","side":"right"},{"sibling":"3107b9d4dbf9456a39f99de694a4dd4da2c0600f9f8855f125161335fe8810af","side":"left"},{"sibling":"86118ab4500c3055a2af70062751a960423c464405b18ca1c37411bf0ce3f52e","side":"left"},{"sibling":"3a42039065acac6d3e4088ec61d9c116ecf7a26c7b7163d23da8fd0b3362e038","side":"left"},{"sibling":"c044f2bd864a0e8e8af5a7f6e3124def7fc4b4511b2b166ea8f9de321e8d385e","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":244827,"merkle_root":"274e133c6dfa2781a9cfb85337d01cc6b72688ce5e810149f3183e400ffab136","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260624T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-24T05:37:55Z","sig_algorithm":"ed25519","signature":"22ca3cee4de2447b3d281e30e09fe566461996bb7be4d4465f08a3f4cf59cea58f22683a4aa10ef4d5d17a19b03f4212392bfd26f2b51f289f0cdf1042a03800","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_18b97dbe6f10419fb49f0c229eb52a852a0b0c67887944bd91495022de1e6f30"}}