{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_48e4b27be945a5e735ec40976886cdb4d72451888be01656e8fcf4b88e710181","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_48e4b27be945a5e735ec40976886cdb4d72451888be01656e8fcf4b88e710181","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"11bf87374c53e8a6657a43144886d80d830a5a8c26d96cd7d6e824457c63cf56","published":"Mon, 25 May 2026 00:00:00 -0400","receipt_hash":"11bf87374c53e8a6657a43144886d80d830a5a8c26d96cd7d6e824457c63cf56","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"11bf87374c53e8a6657a43144886d80d830a5a8c26d96cd7d6e824457c63cf56","observed_at":"2026-05-25T04:43:48.841018Z","parent_run_hash":"56713422f06ad87427cd8cdcdb1ed341feb016b33c37198d9b168328d1df15fb","published":"Mon, 25 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.23262v1 Announce Type: new \nAbstract: The development of LLM agents has led to a growing body of work on knowledge-work AI, including coding, research, and healthcare. However, current knowledge-work evaluation and benchmark design still largely follow the logic of traditional NLP tasks. As a result, higher benchmark performance does not reliably show that a system can carry out knowledge work in real-world deployment settings. This paper contributes a three-step approach for making explicit how benchmarked tasks represent the work claims attached to their scores: defining the work activity under evaluation, specifying the tested setting, and scoring the appropriate work product. We review work studies showing that knowledge work is organized through roles and responsibilities, local materials and tools, and artifacts that must remain usable in downstream workflows. We then translate these concerns into benchmark design and reporting guidance, covering how tasks should be ma","title":"Design and Report Benchmarks for Knowledge Work","url":"https://arxiv.org/abs/2605.23262","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.23262v1 Announce Type: new \nAbstract: The development of LLM agents has led to a growing body of work on knowledge-work AI, including coding, research, and healthcare. However, current knowledge-work evaluation and benchmark design still largely follow the logic of traditional NLP tasks. As a result, higher benchmark performance does not reliably show that a system can carry out knowledge work in real-world deployment settings. This paper contributes a three-step approach for making explicit how benchmarked tasks represent the work claims attached to their scores: defining the work activity under evaluation, specifying the tested setting, and scoring the appropriate work product. We review work studies showing that knowledge work is organized through roles and responsibilities, local materials and tools, and artifacts that must remain usable in downstream workflows. We then translate these concerns into benchmark design and reporting guidance, covering how tasks should be ma","title":"Design and Report Benchmarks for Knowledge Work","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-25T04:43:48Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.23262"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ad4f8a5bd929d2c8e740353a2097d26301f178526a7135a9dadc55bc763e22bfdb2bb8666ad3a7a11fa105ec55cc15449a6ed2f3c9bcad04a5a348564fadda09","signer":"crovia.substrate","subject":{"observed_at":"2026-05-25T04:43:48Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.23262"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"0475ec260f6eb7f3e0fcaf2711982d1990c8445c56e21faa20e2c536677f6a1b","leaf_index":149511,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"4829a470b4b472b0bf3ef3075212426443e188a19368c6985e0a262a0203ce85","side":"left"},{"sibling":"fb1ef5114e2d7812fcb1e47f2efaa3a3c38cae4c1cbb88269d0e19ba0ffd0c49","side":"left"},{"sibling":"108760ea07a053b720810e59ce2ca8c9b3b1a0c98296d0e336a4c222512f550c","side":"left"},{"sibling":"82c22cba13f4a6d4c59dbfeea714c6b37455cd5ee150626bb109b9fe1cdd2f70","side":"right"},{"sibling":"f2249d652df4e3f02af30534b3c51a67de7f3b1a13343680fbbee398aa7c1c83","side":"right"},{"sibling":"6394485477f581e2b46096781b98e87bd7144b0cb4ba48e2ac74aa9b42b595b5","side":"right"},{"sibling":"41d4554111aa829261efa3223e3f2b1b1fb1b33ad3d07b6e2fc8da8c0728c9dc","side":"right"},{"sibling":"d1061d17b49728c47097edde7c7b5b412447bff3df4a459862df4fa81f21f616","side":"right"},{"sibling":"e044acc52c31efe134c902c984299ea3452b42df993d096ed61ad8be9d530bbc","side":"right"},{"sibling":"6be461ecf12dedf98a31921aa7b5c32d4a32df6897e0f697b8bf1a4ba3e2d324","side":"right"},{"sibling":"c2861a8cef3eb66bf2726aa377c24a6bf6e8b7489dcd0e870b61d62a35ccadfb","side":"right"},{"sibling":"134949308b15cffd6792ee2cf678119af34d69a64764d7c89cd47573c94e1cda","side":"left"},{"sibling":"debbc1a6232ea7970009b91bdc2345041b2de5ac59551f955855d579001e512c","side":"right"},{"sibling":"216869846f40bd905626884f58cb67b9019e488b3656946a7808c436ab7339ac","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"e734b0c6d6d13acb5f4cf00367b3913e5bbf1aa717377aa4e4820c6de24b7673","side":"right"},{"sibling":"4bbb7f78e96179bb9cdd06d7207b66b3503ab68e4880438263025b04ebe8f7ec","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":149835,"merkle_root":"51484a548bc0cf3d178eec868e98935c87f0aa83369143b7c96b048585e5ba01","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260525T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-25T05:37:33Z","sig_algorithm":"ed25519","signature":"99a657e53a015ace0bade5d7fbb3f60f950b6d1e1128251a63afcc454a492422cae5efdc0c18503b4f7e62f371bd48f2889be78d9048c4681d6c1e73f754d705","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_48e4b27be945a5e735ec40976886cdb4d72451888be01656e8fcf4b88e710181"}}