{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_2e50c0ebf40387d4ad5e6ce15bb93f63278dcf27d6192544f97423ddc5363c93","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_2e50c0ebf40387d4ad5e6ce15bb93f63278dcf27d6192544f97423ddc5363c93","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e190836cc402f56952a2a381ec65f350161cabef1e85e17cbb90f11b02527e73","published":"Fri, 24 Jul 2026 00:00:00 -0400","receipt_hash":"e190836cc402f56952a2a381ec65f350161cabef1e85e17cbb90f11b02527e73","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e190836cc402f56952a2a381ec65f350161cabef1e85e17cbb90f11b02527e73","observed_at":"2026-07-24T04:43:08.456021Z","parent_run_hash":"b018378f86139a28e6209ec008b31c1282cd1b5c1632dbd43b054c17aa88ab96","published":"Fri, 24 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.21262v2 Announce Type: replace \nAbstract: Reinforcement learning for multi-step LLM agents often relies on scalar rewards that indicate success but cannot explain why a trajectory is good or bad. Rubric-based rewards improve interpretability through natural-language criteria, but existing methods share two limitations: they score at the trajectory level, offering no guidance for individual steps; and their scorer is closed-source and static, so it cannot adapt as the agent evolves during training. We propose ARCO (Adaptive Rubric CO-evolution), which generates a per-step rubric and predicts a rubric-conditioned step-level reward for each action, and continually updates this rubric model on on-policy rollouts so that its criteria and scores co-evolve with the agent's improving behavior. Across HotpotQA, 2WikiMultiHopQA, and MuSiQue with two open-source backbones, ARCO achieves the highest EM in all settings over outcome-, rubric-, and process-reward baselines, and analyses sh","title":"ARCO: Adaptive Rubrics with Co-Evolution for Multi-Step LLM-Based Agents","url":"https://arxiv.org/abs/2606.21262","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.21262v2 Announce Type: replace \nAbstract: Reinforcement learning for multi-step LLM agents often relies on scalar rewards that indicate success but cannot explain why a trajectory is good or bad. Rubric-based rewards improve interpretability through natural-language criteria, but existing methods share two limitations: they score at the trajectory level, offering no guidance for individual steps; and their scorer is closed-source and static, so it cannot adapt as the agent evolves during training. We propose ARCO (Adaptive Rubric CO-evolution), which generates a per-step rubric and predicts a rubric-conditioned step-level reward for each action, and continually updates this rubric model on on-policy rollouts so that its criteria and scores co-evolve with the agent's improving behavior. Across HotpotQA, 2WikiMultiHopQA, and MuSiQue with two open-source backbones, ARCO achieves the highest EM in all settings over outcome-, rubric-, and process-reward baselines, and analyses sh","title":"ARCO: Adaptive Rubrics with Co-Evolution for Multi-Step LLM-Based Agents","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-24T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.21262"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ed68f80bf181f6feb43bc186d239454baefed035664d7f02c790552a13a0b5ce3ddc571908cdc561c215f2cbb75c4a144e6ce2e28223b477c15d0ee0c1d0be04","signer":"crovia.substrate","subject":{"observed_at":"2026-07-24T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.21262"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"89a21a7f5f3fcb2aaaa9fa41b45ff0096bd4273b6643a0bb28f0f06b12710123","leaf_index":347222,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"a22588905f38223eb8813e97bef96e95904bcc917cd1ee81e3c1873eb706f4eb","side":"right"},{"sibling":"03ec0ae872240bbde1885e4979d52d3a7fcb37bf954268c1d0b2e9dd97ea86f8","side":"left"},{"sibling":"035e25d26466620a15a3436139b4eb761f4c7b5432988d0aa86ab376c69c6a0b","side":"left"},{"sibling":"23725355552aef9f43d18bee7fa4a6c15652973a472cfdad6315abd11d12e661","side":"right"},{"sibling":"dbb495c663955cbcf6e9993cd64f61444ca69fd2c1ae1ded3cd67f1a3ca90ef9","side":"left"},{"sibling":"df047800e24edab2e706d2da8a55af57b632bfb1e158709b284e530c20f38633","side":"right"},{"sibling":"22949f5b6c661121ddf62ce35bcf81cd2664e41e1c6aeed348c449bca292ce6b","side":"left"},{"sibling":"aa4b293ba10895bb7f8b28f0f04360b53a8fc5a7523d9a560dbb7b29d003643e","side":"right"},{"sibling":"759f431c97765cb7fb12d38ee64fd6a87076abf1b63c87d7c8c172d57b29084d","side":"right"},{"sibling":"b9cb83ecb59812b3b9270c365951fa79dead288c3923b6cf1f83ffb911b27405","side":"right"},{"sibling":"e6c9083cd0939b38f691f619de2054068654e9c77f7c7ab051e0d48b77339e5d","side":"left"},{"sibling":"d3139af8c5ce235438e1c69e4b7afa44ba09129fd86674968434f23e546f423e","side":"left"},{"sibling":"cb89775a838ee16d10fc8da3213420c2012b4d96e8d55cd49939b0887a4b92d3","side":"right"},{"sibling":"252d30ea8052c3bb6b40bc5cc29fc9b9725343d212f84c08fbbae4215a125b00","side":"right"},{"sibling":"f3e45bceed774d2402fa45d41ff5190f295823bd2f216eb90157884150034693","side":"left"},{"sibling":"3cfa2102c0224815c6f3bf73e6710e24103f43f7bf5da1ca2abad1416d9c0890","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"e871fd7edf9b2ad89bce1609a028f5225eea4d14372169bac242420830f86530","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":347413,"merkle_root":"9efe042c5dd6583dfd3b6a58fbfc289807f60bcf2bd2927f10488a54a8ba11fc","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260724T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-24T05:38:42Z","sig_algorithm":"ed25519","signature":"8633c55f558d42994850505218b2862c6134bad2b1c80d4b80736c2fd3ea7a19690ca3498727a8adbdc47791176128a8d64bef0883b888db809f0477355bd00c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_2e50c0ebf40387d4ad5e6ce15bb93f63278dcf27d6192544f97423ddc5363c93"}}