{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_829b8607a7ff741f84a0d4bf3b5eb1388d6cf83550d19364e75285e25cfc5db0","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_829b8607a7ff741f84a0d4bf3b5eb1388d6cf83550d19364e75285e25cfc5db0","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"18464335efd81e9e8d36bed11e0ab8a4647e6723b2cbbedbb7b4837ff37119e0","published":"Tue, 28 Jul 2026 00:00:00 -0400","receipt_hash":"18464335efd81e9e8d36bed11e0ab8a4647e6723b2cbbedbb7b4837ff37119e0","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"18464335efd81e9e8d36bed11e0ab8a4647e6723b2cbbedbb7b4837ff37119e0","observed_at":"2026-07-28T04:43:08.282317Z","parent_run_hash":"23a1ef85134515049ced29518443d084afc46fd7c967741e6c6acdbdbbf29939","published":"Tue, 28 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2509.02594v3 Announce Type: replace-cross \nAbstract: Evaluating large language models (LLMs) on their ability to generate high-quality, accurate, situationally aware answers to clinical questions requires going beyond conventional benchmarks to assess how these systems behave in complex, high-stakes clinical scenarios. Traditional evaluations are often limited to multiple-choice questions that fail to capture essential competencies such as contextual reasoning, contextual awareness, and uncertainty handling.\n  To address these limitations, we evaluate our agentic RAG-based clinical support assistant, DR. INFO, using HealthBench, a rubric-driven benchmark composed of open-ended, expert-annotated health conversations. On the Hard subset of 1,000 challenging examples, DR. INFO achieves a HealthBench Hard score of 0.68, outperforming leading frontier LLMs including the GPT-5 model family (GPT-5: 0.46, GPT-5.2: 0.42, GPT-5.1: 0.40), Grok 3 (0.23), Gemini 2.5 Pro (0.19), and Claude 3.7","title":"OpenAIs HealthBench in Action: Evaluating an LLM-Based Medical Assistant on Realistic Clinical Queries","url":"https://arxiv.org/abs/2509.02594","vendor":"arxiv_cs_ai"},"summary":"arXiv:2509.02594v3 Announce Type: replace-cross \nAbstract: Evaluating large language models (LLMs) on their ability to generate high-quality, accurate, situationally aware answers to clinical questions requires going beyond conventional benchmarks to assess how these systems behave in complex, high-stakes clinical scenarios. Traditional evaluations are often limited to multiple-choice questions that fail to capture essential competencies such as contextual reasoning, contextual awareness, and uncertainty handling.\n  To address these limitations, we evaluate our agentic RAG-based clinical support assistant, DR. INFO, using HealthBench, a rubric-driven benchmark composed of open-ended, expert-annotated health conversations. On the Hard subset of 1,000 challenging examples, DR. INFO achieves a HealthBench Hard score of 0.68, outperforming leading frontier LLMs including the GPT-5 model family (GPT-5: 0.46, GPT-5.2: 0.42, GPT-5.1: 0.40), Grok 3 (0.23), Gemini 2.5 Pro (0.19), and Claude 3.7","title":"OpenAIs HealthBench in Action: Evaluating an LLM-Based Medical Assistant on Realistic Clinical Queries","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-28T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2509.02594"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:2fb605a8f1c0671fe349d81ff02b2020515969d5c96b20a4467e9cca44dcd3746fb46a4df0ea77d22eb445a0c04e22ca8686d3195188ea01a501f537c77c3003","signer":"crovia.substrate","subject":{"observed_at":"2026-07-28T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2509.02594"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"5d6c2e6f2860134c7025d327073c9000ff535aadf3af9f1a6ce926b531a8c8c0","leaf_index":360784,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1eadb981eeb2016222be1a0650b8ae14f1e963f085f06d6c5a7216b39838898a","side":"right"},{"sibling":"68657669ef892374fb9bc3d352fd1b3ddd439dd1f74bc9350f3da82fe1e31070","side":"right"},{"sibling":"9aef4a650e909a73bd1a32e931a92a09385ad0f4e5cadccd162cc5d167e5194b","side":"right"},{"sibling":"c062f24ac40c63354c96afdc7539004b9659ce36773f96b11528c621f4f00ad6","side":"right"},{"sibling":"b9bb11171fd35649bbec8b183eecf18541116b1c9d4ebf1bfbbbe65f69afac33","side":"left"},{"sibling":"d5cd56730e6e2232d8eefa52f35b8a56d78b86d619fa7eb71110315f04a1d6cc","side":"right"},{"sibling":"edee58372a639bd2a7ffe555c89a5f76b628b6fb4ff60a0320a74d44c214bdd6","side":"left"},{"sibling":"80654268ff95f1daa72785465b969eee9ea2b22da6f8dfc0d2cf6148d4718a52","side":"right"},{"sibling":"f05fa4d2088913dc5410109b6c9f2c28b5e413f06a98f46cd68973c567c3b416","side":"left"},{"sibling":"fe03c6d0b083c4097049d7fc6d7d08dfdb05d9c198b2786336689712d66eabf0","side":"right"},{"sibling":"590ea76fdfc1b9e8072055c378be3182f916709e8d06bd955232e650a7188b86","side":"right"},{"sibling":"d92781c59301ffd5bfb0bad75d9fdf6d73879f715518149d383f7362213e0daf","side":"right"},{"sibling":"f315303d4402b57497416d48eb4c4bb50405b40862d41c7caf318cb3d29c5237","side":"right"},{"sibling":"36973eb5f586cd67e0c0dc055dd87e734aa544c35d4400d2c9f932f8ef8fb27f","side":"right"},{"sibling":"e39f7900355489c4718b21cc2d3d06382e1d2f22b864d10be9ebc9d498b279a1","side":"right"},{"sibling":"1f9a970b25dd938c98cabc9e0a55c5a6f46292b9fa37cc89d90ef0cbb1e05a8c","side":"left"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"3b50864499c874394ea0928567747666eaf59b01380e46cd52164ec5acec0f71","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":361008,"merkle_root":"3065e8369ea437c06beba806dc4e4bb159979adeb21fe632242c1906a7204647","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260728T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-28T05:38:48Z","sig_algorithm":"ed25519","signature":"9141644407577a82611c1579110f667de2d46dc6b93c6322edf26f4c3056ea99f0e56502853908e30d87c38bcf95eb6e0ab5130525aa51505bd6f61938120609","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_829b8607a7ff741f84a0d4bf3b5eb1388d6cf83550d19364e75285e25cfc5db0"}}