{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1a2c80e1ec98ade969c114050fc8bec174e271cf48981dc3621c12773f6a2bbf","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1a2c80e1ec98ade969c114050fc8bec174e271cf48981dc3621c12773f6a2bbf","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"ec4c5c239c10810069a5d1828c6a56622f3485f4111c41f4bf66b73b74f3b998","published":"Thu, 23 Jul 2026 00:00:00 -0400","receipt_hash":"ec4c5c239c10810069a5d1828c6a56622f3485f4111c41f4bf66b73b74f3b998","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"ec4c5c239c10810069a5d1828c6a56622f3485f4111c41f4bf66b73b74f3b998","observed_at":"2026-07-23T04:43:18.446221Z","parent_run_hash":"b3f5e4095688e31d15c25de2607bca42111343bdfdac967d48e47905415ddea0","published":"Thu, 23 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2601.06487v3 Announce Type: replace-cross \nAbstract: Reinforcement learning has substantially improved the performance of LLM agents on tasks with verifiable outcomes, but it still struggles on open-ended agent tasks with vast solution spaces (e.g., complex travel planning). Due to the absence of objective ground-truth for these tasks, current RL algorithms largely rely on reward models that assign scalar scores to individual responses. We contend that such pointwise scoring suffers from an inherent discrimination collapse: the reward model struggles to distinguish subtle advantages among different trajectories, resulting in scores within a group being compressed into a narrow range. Consequently, the effective reward signal becomes dominated by noise from the reward model, leading to optimization stagnation. To address this, we propose ArenaRL, a reinforcement learning paradigm that shifts from pointwise scalar scoring to intra-group relative ranking. ArenaRL introduces a proces","title":"ArenaRL: Scaling RL for Open-Ended Agents via Tournament-based Relative Ranking","url":"https://arxiv.org/abs/2601.06487","vendor":"arxiv_cs_ai"},"summary":"arXiv:2601.06487v3 Announce Type: replace-cross \nAbstract: Reinforcement learning has substantially improved the performance of LLM agents on tasks with verifiable outcomes, but it still struggles on open-ended agent tasks with vast solution spaces (e.g., complex travel planning). Due to the absence of objective ground-truth for these tasks, current RL algorithms largely rely on reward models that assign scalar scores to individual responses. We contend that such pointwise scoring suffers from an inherent discrimination collapse: the reward model struggles to distinguish subtle advantages among different trajectories, resulting in scores within a group being compressed into a narrow range. Consequently, the effective reward signal becomes dominated by noise from the reward model, leading to optimization stagnation. To address this, we propose ArenaRL, a reinforcement learning paradigm that shifts from pointwise scalar scoring to intra-group relative ranking. ArenaRL introduces a proces","title":"ArenaRL: Scaling RL for Open-Ended Agents via Tournament-based Relative Ranking","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-23T04:43:18Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2601.06487"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1d638a09faffbd47c98107232fed3b087fc41a9fb4ada29d8713e39e1f7f4a425de8cfd81cb93ce6fa7320cb5b7c44e248abb47d9535742f26f4c06971c4cf09","signer":"crovia.substrate","subject":{"observed_at":"2026-07-23T04:43:18Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2601.06487"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f614cf4c5414899663e5bd520a4fa5caf90a6e91e38ec4e5f57c9c357e4de6a6","leaf_index":343783,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7e373cd7cbe8c27b36a7f055065a74776e6dd87d3a98969a5ea4f3dbc00ed389","side":"left"},{"sibling":"9e35ea0f8c2eafa7874cf84a660f24d44e272f04916c4cb03efd9bda40e82d1b","side":"left"},{"sibling":"41dcfff50f10a5455533fe4b9658b290ebe0d605a1cc16f43be08a109e5084fc","side":"left"},{"sibling":"3fc80d68be74125e490a9a1ab803b141f981e7fe8e1d9996d8f974bd2a9d03e8","side":"right"},{"sibling":"c423429d1a8c3090b58f8897f311a3b03f9f60ac7fb988ff530039a6c6e7c129","side":"right"},{"sibling":"8c4abba7f73d017263c7cfa7cf00e7dad2248eccd24b930d64f9284ce5a8897d","side":"left"},{"sibling":"84e2bd69f70ff0e814f453c01a295fde8d94bdb3824a1424ab4fd97be4fa4cba","side":"left"},{"sibling":"fbe8bc0f33cd1a297ce2f845b101ed1107ebb67f2c2577c28ed032c1046d3b97","side":"left"},{"sibling":"6ecdcc1e2fb6ab44777fffbb7c8297d723018c63aac9d90c7a1bbebe728d84cf","side":"right"},{"sibling":"93d7d8e0e882d05b2a15bb707a824979a0427907eb47e687c712906674a0d345","side":"left"},{"sibling":"92219a3ef58cd145d94f071b0b9396cec3707812b02a8c7c63f2d0e22340552b","side":"left"},{"sibling":"4eb402d67bd4bf583b0434363061166fe259c34cc6c42adb32dcfbff0a9f5767","side":"left"},{"sibling":"2dd9cb2521044ee7c6b74f2315e0a0253b8df0d04a7b810bbbbe7da5a9788769","side":"left"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"941f71d7ce3990a507b8f485de3f872a51d0e9a8c59405d76c2ae6f4a6af494a","side":"right"},{"sibling":"6281b6f7a93c44e3c4895bc65cfcb6f2be24dd725f4f46540ec022a6e215f4e8","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"6baede22892163664bb2e4d92cf75e6b290761533a6c39491c3afd89bb3a0252","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":343944,"merkle_root":"afab58f71597d393a7a61b857dac2eacb72fd1c04cd1432c2622d7b19309dffd","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260723T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-23T05:38:39Z","sig_algorithm":"ed25519","signature":"1a58fdc6367d0c86f83d2385be748e0800a64bcf9987c92fadca53deda009cd390d2070302c2b6e44841febfd864637c323ba12c7a1e9ae58fdf2a0cf546ac01","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1a2c80e1ec98ade969c114050fc8bec174e271cf48981dc3621c12773f6a2bbf"}}