{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_bf5407ca6b4c86751743cf73853ad63dcb20bae9807afd7508bdb62cf9f065c6","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_bf5407ca6b4c86751743cf73853ad63dcb20bae9807afd7508bdb62cf9f065c6","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a32094155476e4a6bdc65900fcfa8a061535300cc28ad05d7827a29ebf857d21","published":"Thu, 28 May 2026 00:00:00 -0400","receipt_hash":"a32094155476e4a6bdc65900fcfa8a061535300cc28ad05d7827a29ebf857d21","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a32094155476e4a6bdc65900fcfa8a061535300cc28ad05d7827a29ebf857d21","observed_at":"2026-05-28T04:43:38.862500Z","parent_run_hash":"58f8b4a134069e0a15ea3949252489597eb86dd27c9ca3fb15c6fb838ce49ef3","published":"Thu, 28 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.27898v1 Announce Type: new \nAbstract: As LLMs are increasingly deployed as agents, reliable assessment of their agentic capabilities has become essential. However, reported benchmark scores often jointly reflect model capability and the implementation choices each benchmark is packaged with, making cross-benchmark results difficult to interpret as clean measurements of the underlying model. In this work, we present a unified framework for the fair evaluation of LLM agentic capabilities. Driven by a unified configuration system, the framework integrates diverse benchmarks into a standardized instruction--tool--environment format, executes agents through a fixed ReAct-style architecture within a controllable sandbox, and provides an optional offline setting that replaces volatile live environments with curated snapshots, so that framework effects and environment effects can be analyzed separately. Building on this, we unify the evaluation methodology under each benchmark's ori","title":"A Unified Framework for the Evaluation of LLM Agentic Capabilities","url":"https://arxiv.org/abs/2605.27898","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.27898v1 Announce Type: new \nAbstract: As LLMs are increasingly deployed as agents, reliable assessment of their agentic capabilities has become essential. However, reported benchmark scores often jointly reflect model capability and the implementation choices each benchmark is packaged with, making cross-benchmark results difficult to interpret as clean measurements of the underlying model. In this work, we present a unified framework for the fair evaluation of LLM agentic capabilities. Driven by a unified configuration system, the framework integrates diverse benchmarks into a standardized instruction--tool--environment format, executes agents through a fixed ReAct-style architecture within a controllable sandbox, and provides an optional offline setting that replaces volatile live environments with curated snapshots, so that framework effects and environment effects can be analyzed separately. Building on this, we unify the evaluation methodology under each benchmark's ori","title":"A Unified Framework for the Evaluation of LLM Agentic Capabilities","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-28T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.27898"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ca929feadfdac5d5558b524eb28746dcc52a213ca28a55c57ceac60a80b8241ce0765441ea96fceb85a701a35115ad25de7835af258d417ecc61b47faa5fd604","signer":"crovia.substrate","subject":{"observed_at":"2026-05-28T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.27898"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"31b56bab6d38414cf2a5199457afb82e29d7053d03fe26a122441879c3483058","leaf_index":155611,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f802272c67fd25b23abfc278080f8a370bbaef994c1039fb18530ef342f57443","side":"left"},{"sibling":"31b97e0a772c2ef21eca76c2788196406024b8324babf79e8fa0f9408a451ef2","side":"left"},{"sibling":"212fa5afd846952dab79e4416a08c9821754eb0d4f9bbe07b895f186c8008d49","side":"right"},{"sibling":"8e1da781a41050afc2962abd670bf79a21c12bdad3ec09051399428ae8b2fed8","side":"left"},{"sibling":"c8c21e6ef77478e42625acc377a4062ee6ad593745301a1f6d90936e6d94e475","side":"left"},{"sibling":"9271164ea0f2c427fba10eb695cb98be464e920bb5d4d839cf341ff315753f23","side":"right"},{"sibling":"7252c2f7874601f6b2d750e8a0aa6cc72c6924d6614fe237cd5c97d2026c5c17","side":"left"},{"sibling":"70e73f197ecea5d12ff2f9cf2ca33087110a782ea956c539b3058422ada4b2bd","side":"left"},{"sibling":"8c9c7b64ca041ef0029c35e402972d0bebed706fe997b9c587da6f5c811dcce1","side":"left"},{"sibling":"aa7a574eaea239ab8851da225206477d62a6ea5a15b65834f22d1c10288348e7","side":"left"},{"sibling":"7fcba2ee8232873e5d864902f28968d2be38867751c238f238f629eddb98b226","side":"left"},{"sibling":"816f233274bb10f5a122aac086a0c8c697b78fec67a4af55190bb596b7506fab","side":"left"},{"sibling":"d415e6939aee710631f5062799379b547d2c3e3d9a68f263bbb5a693285ab2ca","side":"left"},{"sibling":"d422d38e0ffe6e849b6ce3259d90c9d36497b4d7391a003e15591668c79028a7","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"7b6f0bea4291a5e63574dfca9aa0f9756f450c9478c3d07474d39a7ababb51f9","side":"right"},{"sibling":"1d39fe14b21e2ebbfb87e882423b24ee9469eae1e4c77af5b799ac4db9537467","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":156177,"merkle_root":"c5705a0243d16afd8b1ebfd731b7aa304079c442c2a7906493c5bbed374c69ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260528T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-28T05:37:36Z","sig_algorithm":"ed25519","signature":"f087e13febc8bb6a2e0812610de64cebc65be92915518d9c4b230799c3b161839b04c4eb1b02741f938f35545a76ab76555b04c782bdc2f9a44852d171d65909","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_bf5407ca6b4c86751743cf73853ad63dcb20bae9807afd7508bdb62cf9f065c6"}}