{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_679d5acf5cea6244dbed3a92681ea550bccd6ec076758d0796af1d8c19d76c70","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_679d5acf5cea6244dbed3a92681ea550bccd6ec076758d0796af1d8c19d76c70","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e254b692a45359416210488166fad2164df71fcf09fb2820a6899421f4fe12c5","published":"Wed, 10 Jun 2026 00:00:00 -0400","receipt_hash":"e254b692a45359416210488166fad2164df71fcf09fb2820a6899421f4fe12c5","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e254b692a45359416210488166fad2164df71fcf09fb2820a6899421f4fe12c5","observed_at":"2026-06-10T04:43:37.461885Z","parent_run_hash":"23aff1a6f676ba7ca33f70f4ddfae1dd282fb86104d577ce9be510d81a94c5dc","published":"Wed, 10 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.10394v1 Announce Type: new \nAbstract: Large language models are increasingly used to power personal agents for everyday applications, but evaluating these agents remains a challenge. Existing benchmarks still rely on sandboxed artifacts, static task design, and coarse scoring, which hinder scalability and limit progress toward reliable personal-agent evaluation. This paper introduces STAGE-Claw, an automated framework for building and evaluating realistic personal-agent scenarios in state-based personal-computing environments. Given a task hint, STAGE-Claw automatically creates and validates a realistic benchmark task with its environment, task prompts, ground truth, and related verification programs. Agents are then evaluated in realistic operating environments, where performance is measured by the correctness of the final system state rather than only the textual response. Using STAGE-Claw, this paper creates a benchmark with 40 challenging real scenario agent tasks, evalu","title":"STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios","url":"https://arxiv.org/abs/2606.10394","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.10394v1 Announce Type: new \nAbstract: Large language models are increasingly used to power personal agents for everyday applications, but evaluating these agents remains a challenge. Existing benchmarks still rely on sandboxed artifacts, static task design, and coarse scoring, which hinder scalability and limit progress toward reliable personal-agent evaluation. This paper introduces STAGE-Claw, an automated framework for building and evaluating realistic personal-agent scenarios in state-based personal-computing environments. Given a task hint, STAGE-Claw automatically creates and validates a realistic benchmark task with its environment, task prompts, ground truth, and related verification programs. Agents are then evaluated in realistic operating environments, where performance is measured by the correctness of the final system state rather than only the textual response. Using STAGE-Claw, this paper creates a benchmark with 40 challenging real scenario agent tasks, evalu","title":"STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-10T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.10394"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a5a8311f4b0f8b00fdd1280ce660e1054740bf7e215ec124caa68e5cf3e89ec0b8e3a26b4b70be2954baa10cb387366f3cdd8585fa8c1deec30b3ed4382e3006","signer":"crovia.substrate","subject":{"observed_at":"2026-06-10T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.10394"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"74630646b60e147e6c1d6b85e271a89bad20347103c2f1c4eafacb326d239d0a","leaf_index":225857,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7bfe8f6e606d32d5b2eaa9dde24764dbfa97a65d7ab1ea80aa7f907181cec364","side":"left"},{"sibling":"e869ad599ebe85a9df23beb0397a98854e53d050c4bf59e348deab8a52eb731a","side":"right"},{"sibling":"0a24f0b64b9e9b3a624c1eaa2612f47207958a0c7a90cb1a804b4bdb508550a2","side":"right"},{"sibling":"361426d53a70dfbbb8bca46da66816b8d7598361965ad77684591d7c862eda65","side":"right"},{"sibling":"3b813cab10ce7a66b743bb8fd37df43bb891ecac0c984731c71f5d6c431f6bad","side":"right"},{"sibling":"2e627a04aead5091204cc45a543ab572843c0df27707aa2624afc983f9387e00","side":"right"},{"sibling":"b2b764628ccd5a7c6a8404965429a36513804c4f359049b6a0f70892a8a53490","side":"left"},{"sibling":"b42fc5e6a220ca97d9fe75065fb0dafd72b915ddda12e63083678d3813da9eee","side":"right"},{"sibling":"ac698c3a6027f35b513fe892f166e70344e855625238cbb581d0bf06c131db80","side":"right"},{"sibling":"a4d17aefe58175050dc159af6246658fcf1c9f3ed57aacf1b350fc3261de4e69","side":"left"},{"sibling":"280b980aa0c7756b0b0cb22658f26466d36f0e70fbc3312cd2311d9898e30b8f","side":"right"},{"sibling":"c98954d4b658b1dda60fe52576fcf9bf21a2d49c67fb63f8c30f16ab5f721938","side":"right"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_679d5acf5cea6244dbed3a92681ea550bccd6ec076758d0796af1d8c19d76c70"}}