{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b4c463743dc4e197a17038c6415371feedd0fc0452d98fa44d687114e8f1c0e8","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b4c463743dc4e197a17038c6415371feedd0fc0452d98fa44d687114e8f1c0e8","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e3c5d6850995e96ab54f1debe7f15544d031a481e35aeaceff6bb070057c2cbb","published":"Wed, 03 Jun 2026 00:00:00 -0400","receipt_hash":"e3c5d6850995e96ab54f1debe7f15544d031a481e35aeaceff6bb070057c2cbb","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e3c5d6850995e96ab54f1debe7f15544d031a481e35aeaceff6bb070057c2cbb","observed_at":"2026-06-03T04:43:57.136784Z","parent_run_hash":"62ae9c8eda846b00bc49666345b338d00756c5203438666c4c1fc694cc364b84","published":"Wed, 03 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.03198v1 Announce Type: cross \nAbstract: Clinical AI evaluation increasingly delegates scoring to large language models (LLMs) acting as AI raters, yet their scoring behavior across evaluation conditions has not been quantitatively characterized. We address this gap through a factorial study of AI rater behavior in adult type 2 diabetes (T2D) pharmacotherapy at 12-month outpatient follow-up, a clinical task involving complex decision-making operationalized across seven evaluation questions. Four open-source LLMs served simultaneously as clinical decision support system (CDSS) models and AI raters. Each CDSS output was scored under two scoring protocols: a rubric-anchored Gold Rubric (GR) protocol incorporating a patient-specific rubric, and a rubric-free Non Gold Rubric (Non-GR) protocol. Linear mixed effects models crossed the scoring protocol factor with five design factors -- CDSS model, CDSS prompt configuration (document-referenced generation [DRG] vs.\\ Baseline), rater ","title":"AI Rater Discrimination Depends on Scoring Protocol in Complex Clinical Decision-Making","url":"https://arxiv.org/abs/2606.03198","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.03198v1 Announce Type: cross \nAbstract: Clinical AI evaluation increasingly delegates scoring to large language models (LLMs) acting as AI raters, yet their scoring behavior across evaluation conditions has not been quantitatively characterized. We address this gap through a factorial study of AI rater behavior in adult type 2 diabetes (T2D) pharmacotherapy at 12-month outpatient follow-up, a clinical task involving complex decision-making operationalized across seven evaluation questions. Four open-source LLMs served simultaneously as clinical decision support system (CDSS) models and AI raters. Each CDSS output was scored under two scoring protocols: a rubric-anchored Gold Rubric (GR) protocol incorporating a patient-specific rubric, and a rubric-free Non Gold Rubric (Non-GR) protocol. Linear mixed effects models crossed the scoring protocol factor with five design factors -- CDSS model, CDSS prompt configuration (document-referenced generation [DRG] vs.\\ Baseline), rater ","title":"AI Rater Discrimination Depends on Scoring Protocol in Complex Clinical Decision-Making","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-03T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.03198"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:dbba0498a174d4e05fcb605876ff3cfd7f26527f0a74830b647eebc33dcd9ac5977e430a28b82e0ce2a0c9a4ef5aad2b1dabb22aad851d6255e2dd339b39580d","signer":"crovia.substrate","subject":{"observed_at":"2026-06-03T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.03198"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"9d309b0f44b1cf8ff8004457a614200f6bb3225115aee5a75f2b2403f6b60f7e","leaf_index":209189,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"9834c7640ebfa3e8438d7af56d6e08b6f09974e2b876db243a238359c90d6df6","side":"left"},{"sibling":"4b96a6802c13af930cc72b3a7045e5afd93fe0ea671ec49684d6ba991f45fec9","side":"right"},{"sibling":"a7a8ba72157e9171cf8d24bd2ebebd9c140531ee21cad6536bfa9d61d9e93e93","side":"left"},{"sibling":"e4e5019838277969ab1d4c17ac4518608d793d3161eaf8d8163ad95d47360c3b","side":"right"},{"sibling":"1ccc3a592317446475b5176643f5fb3a0d50da221bf9169a66e562823ce4f3c8","side":"right"},{"sibling":"f8476278732bb4d41b8c08d522b3f3fda7a02a51bed0d8aa9461d027be7500a1","side":"left"},{"sibling":"40052708582da742bf088b0c30cd584ba95f867aa32f06040dcfa4b5df18f737","side":"right"},{"sibling":"2ef1f1a576d002b7a433f7807e3e3de79248e5d1145c4394fef6eb45b8063603","side":"right"},{"sibling":"2b9867040ec52d22722edc26b204e15651723dcc66d361cc1af11490a854d461","side":"left"},{"sibling":"2abcec6d14b256f82b270a3141886de6129141dec525543607b760f02588a877","side":"right"},{"sibling":"2b6b45743f97ac502854e489ac38a3366f8ae7ede58a2728b45daadbf29e9d03","side":"right"},{"sibling":"a81babbd79ea0da9e030dd7f43bffb6519d214317decd50727bea4e78189d970","side":"right"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"8d3baa674a45fd8bc6d3e8d25298f4bec86c72fa8259d576c330d12955c7b4f7","side":"right"},{"sibling":"e32819d1eff909db08066d1703f2db3f091cddac19378b2c0c625ab11e3fdbc0","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":209569,"merkle_root":"c852efc8ad7dfffc196c71380f79e6398bcaf566974cbae7c9950b0f600d5bc8","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260603T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-03T05:37:57Z","sig_algorithm":"ed25519","signature":"1ca417607effc813341721cfdade0a2d2d4ba96a90d361dbc3e864b6991b90abc326ec41d9ba3e7110cc04adb29d7b8d95abea7ff2ab8c591b2f864ed7270c03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b4c463743dc4e197a17038c6415371feedd0fc0452d98fa44d687114e8f1c0e8"}}