{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_49c1ad6df9ea5c181047454d107ff52b73e4718be3c97102f442dd2fcac39dc4","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_49c1ad6df9ea5c181047454d107ff52b73e4718be3c97102f442dd2fcac39dc4","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"4acd0dc5105cba3e37a842e72a62210fa0394e8eb1149fa9a58bfb92027a2548","published":"Tue, 14 Jul 2026 00:00:00 -0400","receipt_hash":"4acd0dc5105cba3e37a842e72a62210fa0394e8eb1149fa9a58bfb92027a2548","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"4acd0dc5105cba3e37a842e72a62210fa0394e8eb1149fa9a58bfb92027a2548","observed_at":"2026-07-14T04:43:37.834979Z","parent_run_hash":"66b89520a448b8d9fe7d8f602ef38b82b6c95e57f532ce72de51375b41870477","published":"Tue, 14 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.06486v3 Announce Type: replace \nAbstract: Evaluating agentic AI on open-ended professional tasks faces a fundamental dilemma between rigor and flexibility. Static rubrics provide rigorous, reproducible assessment but fail to accommodate diverse valid response strategies, while LLM-as-a-judge approaches adapt to individual responses yet suffer from instability and bias. Human experts address this dilemma by combining domain-grounded principles with dynamic, claim-level assessment. Inspired by this process, we propose \\textbf{JADE}, a two-layer evaluation framework. Layer 1 encodes expert knowledge as a predefined set of evaluation skills, providing stable evaluation criteria. Layer 2 performs report-specific, claim-level evaluation to flexibly assess diverse reasoning strategies, with evidence-dependency gating to invalidate conclusions built on refuted claims. Experiments on BizBench show that JADE improves evaluation stability and reveals critical agent failure modes missed","title":"JADE: Expert-Grounded Dynamic Evaluation for Open-Ended Professional Tasks","url":"https://arxiv.org/abs/2602.06486","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.06486v3 Announce Type: replace \nAbstract: Evaluating agentic AI on open-ended professional tasks faces a fundamental dilemma between rigor and flexibility. Static rubrics provide rigorous, reproducible assessment but fail to accommodate diverse valid response strategies, while LLM-as-a-judge approaches adapt to individual responses yet suffer from instability and bias. Human experts address this dilemma by combining domain-grounded principles with dynamic, claim-level assessment. Inspired by this process, we propose \\textbf{JADE}, a two-layer evaluation framework. Layer 1 encodes expert knowledge as a predefined set of evaluation skills, providing stable evaluation criteria. Layer 2 performs report-specific, claim-level evaluation to flexibly assess diverse reasoning strategies, with evidence-dependency gating to invalidate conclusions built on refuted claims. Experiments on BizBench show that JADE improves evaluation stability and reveals critical agent failure modes missed","title":"JADE: Expert-Grounded Dynamic Evaluation for Open-Ended Professional Tasks","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-14T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.06486"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:5d0d6f23552c4d311c18b1ecaf2e005ba3fd4790bf9b9a25d744721cc02b435456df869946f9ab0ac481ea5227cea7052f89d4831774ac8c5bb7ad96f8225e07","signer":"crovia.substrate","subject":{"observed_at":"2026-07-14T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.06486"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d817a2fb3b80a581b7a8cd046f43d4ea33448b6b1e77461ff24fffc6af2a3a4f","leaf_index":313123,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"a04cf643797520f8aa75f6d68aeced2d302295a3c8fe9168ec7cd8ae231c556c","side":"left"},{"sibling":"68da250974f327818b2c8c89491df7f9dcbbdea5cadc981c29ca3cd582768a3e","side":"left"},{"sibling":"9e83ab38b9b21658e5318c2e59c6a4654814b9f1d9b52c225ee79b3b1c98eba0","side":"right"},{"sibling":"37c07ee1f458143d4786ade9ddda58a4f458f55ca9e384bdae8bf804fd1a37e6","side":"right"},{"sibling":"cc8bf68ae0277149de11f9418c8ca27270cb49121a0e31f2b4c553e855efdfa9","side":"right"},{"sibling":"63c17f44c675a07d319f8cc451137e3f1ec8aa5e4b9afb9f94042cbadea1a6d3","side":"left"},{"sibling":"a833523d950428dc21ca9a8e4dda7f94685e705281058be8f40306936c6e2602","side":"right"},{"sibling":"edf133a890c50dc04a7d212f979f1424434d765ec9a191894fda402b340d9450","side":"right"},{"sibling":"ae8fbfdec46ce02d0a92359c6a034f0439394d13e731da90290f5ca7192e6da4","side":"left"},{"sibling":"8393dc03644000353c5d503b6ec261c022cd2aa85a649223916153bc8af2dc75","side":"left"},{"sibling":"1627ce6908965f2e5fbc7c16bc8e247867478c587014561d0de1359dc2dc0afc","side":"left"},{"sibling":"74e1ba9fd48bdd0f52777dd5b56c10706e73f42629013748eca13ef113cd59db","side":"right"},{"sibling":"682fd39539db5bce908d0ab6f180374728fed6b34a9a8c87c7b4eb5a726e30b8","side":"right"},{"sibling":"184927f71d667ac0c6ebeadb33394626973738b9107c6dc3c0c4949b44acf295","side":"right"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"c4c189d79669979b59d9aa976ee249c54a7e065b4a05bf49463459cc7f37428e","side":"right"},{"sibling":"19475e206bdf2698769a286db2c97c4d3f089741319f9b29a139038c6e511975","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":313381,"merkle_root":"5f7d48580373d0ab0bc6bdf34f26de442f9a86d130946f7ee45addd1a55cb9f4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260714T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-14T05:38:23Z","sig_algorithm":"ed25519","signature":"97364224142ed1b2a8925fed1bfba2e0a8a6a15f8ca0999b934846ad83584a78c7f0f5cd09dbc1c938b35383fcc5c8fa4b723118e30916628879d1c5fd3a9908","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_49c1ad6df9ea5c181047454d107ff52b73e4718be3c97102f442dd2fcac39dc4"}}