{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_167233f9f3e14eeb2506769df240eef216b123b8b8609c5487514db809fc4463","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_167233f9f3e14eeb2506769df240eef216b123b8b8609c5487514db809fc4463","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f62697456321ddefa8057f243a6cc82dd31a4eb979447d9caf3d669d22934ae1","published":"Wed, 08 Jul 2026 00:00:00 -0400","receipt_hash":"f62697456321ddefa8057f243a6cc82dd31a4eb979447d9caf3d669d22934ae1","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f62697456321ddefa8057f243a6cc82dd31a4eb979447d9caf3d669d22934ae1","observed_at":"2026-07-08T04:43:57.834712Z","parent_run_hash":"46ab019f0b0f0bfcde5e14ed7c256069c6fa8c8079b9f87fd3a8a6d9259e3864","published":"Wed, 08 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.19990v2 Announce Type: replace \nAbstract: While RL has become a promising tool for refining world models, existing methods largely rely on conservative rollouts near the training distribution, limiting exploration, behavioral diversity, and richer dynamic discovery. In this work, we challenge this conservative paradigm. We argue that the core limitation is not exploration itself, but the lack of reliable verification strategies to support broader exploration. Without reliable verification, expanded exploration becomes highly susceptible to reward hacking, where policies exploit imperfect rewards without achieving genuine improvement. To evaluate this motivation, we instantiate our method in embodied world models, where physical plausibility, and task completion provide a rigorous testbed for scalable RL under complex dynamics. On the verification side, we introduce Reward as an Agent, an agentic reward framework that actively evaluates generated behaviors to provide robust r","title":"Reward as An Agent for Embodied World Models","url":"https://arxiv.org/abs/2606.19990","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.19990v2 Announce Type: replace \nAbstract: While RL has become a promising tool for refining world models, existing methods largely rely on conservative rollouts near the training distribution, limiting exploration, behavioral diversity, and richer dynamic discovery. In this work, we challenge this conservative paradigm. We argue that the core limitation is not exploration itself, but the lack of reliable verification strategies to support broader exploration. Without reliable verification, expanded exploration becomes highly susceptible to reward hacking, where policies exploit imperfect rewards without achieving genuine improvement. To evaluate this motivation, we instantiate our method in embodied world models, where physical plausibility, and task completion provide a rigorous testbed for scalable RL under complex dynamics. On the verification side, we introduce Reward as an Agent, an agentic reward framework that actively evaluates generated behaviors to provide robust r","title":"Reward as An Agent for Embodied World Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-08T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.19990"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:f87930c69f9fb41d5fee5dd76537678ad118523c1a52a0013481e322b54a8827685c0b973d58e152118f5b36475e5db489c012fa90504b71f54c011146faf003","signer":"crovia.substrate","subject":{"observed_at":"2026-07-08T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.19990"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"865e2bfddb62d7f53ba4c2c280b1a48a06fcfa764d38a9ceb6a76c07878cf67e","leaf_index":292808,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1b52567f6157e6fed81628579d333acedfcd5a7d7777c8d9a2c1ecde5b9e5a3c","side":"right"},{"sibling":"e0a1b05fb7b77ea0a4d012c6bc9b875d430e0a2fc27fcf5ded67333e6ca2514b","side":"right"},{"sibling":"397c3056df5bc1a37d33c08b0640f82f26fb6c3e6cc9ca86c12773cea0581e5b","side":"right"},{"sibling":"4f95a66a062efe7db3de40028aa0733ff80310d6d26a7658a638bd570ea860dd","side":"left"},{"sibling":"321a20155f8045e1d89e7d6d4f5f69c5fafde4faff272dad489b3d34aa525f9d","side":"right"},{"sibling":"c3bcd80d31f44b45ac24d75235b431952a5fbdf53c2994e96b08aa35f6212442","side":"right"},{"sibling":"b68effed96c9c760c6a644ba421552cb75cab0f5269153819ebc6ee58c6e16f1","side":"left"},{"sibling":"6e4b9570930ece9a9ba8eab260493982ccceb0060de1dddb4ef505f9f3d00598","side":"left"},{"sibling":"1515812edbf9903d3d787f20218b8577e2f1fef32592508ce01a3b3ebe75f507","side":"left"},{"sibling":"d4b475beae71e5b4d5ab7e66f7144e7b8e1356fcd32339e49df234b455659fef","side":"left"},{"sibling":"cead64e0790e8871aeea2334210146b5e35fe7e4bc10748acdf76da0c565be05","side":"left"},{"sibling":"d1231ac6e6bd7d6867a9109fbdadede0b1631e97866ddde70f7ac2d52c28e15f","side":"right"},{"sibling":"76855b4804c75c52bf97aa34358950d42d6103cdc1a86be5f0a2c8de4d65c106","side":"left"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"9e75f2ab0ddf2dc9e92af7049244c21b909734ab57906a35dfad2853ca9966e2","side":"right"},{"sibling":"e6cd4cad39a4b6ca6647d1b0ad2db86e57e5fa6240f65966c10093e91140769b","side":"right"},{"sibling":"90a7efc6b94ec8913fbdf03f4927a821b9fb89921d526716f5ee28f015303779","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":292998,"merkle_root":"f533b7efebdfd8fb6ba3e7cc158ee55261fd53a7985f234cfea359170dad4d5a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260708T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-08T05:38:17Z","sig_algorithm":"ed25519","signature":"07ebb10c732bbffb28b55a5db01d5525f6b8ab7ef36e1e0a68c35c96ded99f7040c277edba75eb6b15c77feda30bd5321e31ada6f572b674d06f4f8e24bd2f07","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_167233f9f3e14eeb2506769df240eef216b123b8b8609c5487514db809fc4463"}}