{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_dbf20cb313687e72db75741cdcf2800d697209dc5c6ef20ed40d9de533db16cf","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_dbf20cb313687e72db75741cdcf2800d697209dc5c6ef20ed40d9de533db16cf","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"25d7dd62c9359c1d2cfc113325724faa15f6edad67e14f73b4295001ed1bb6e5","published":"Wed, 03 Jun 2026 00:00:00 -0400","receipt_hash":"25d7dd62c9359c1d2cfc113325724faa15f6edad67e14f73b4295001ed1bb6e5","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"25d7dd62c9359c1d2cfc113325724faa15f6edad67e14f73b4295001ed1bb6e5","observed_at":"2026-06-03T04:43:57.136784Z","parent_run_hash":"62ae9c8eda846b00bc49666345b338d00756c5203438666c4c1fc694cc364b84","published":"Wed, 03 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.23099v2 Announce Type: replace-cross \nAbstract: Evaluating generative AI models is increasingly resource-intensive due to slow inference, expensive raters, and a rapidly growing landscape of models and benchmarks. We propose ProEval, a proactive evaluation framework that leverages transfer learning to efficiently estimate performance and identify failure cases. ProEval employs pre-trained Gaussian Processes (GPs) as surrogates for the performance score function, mapping model inputs to metrics such as the severity of errors or safety violations. By framing performance estimation as Bayesian quadrature (BQ) and failure discovery as superlevel set sampling, we develop uncertainty-aware decision strategies that actively select or synthesize highly informative inputs for testing. Theoretically, we prove that our pre-trained GP-based BQ estimator is unbiased and bounded. Empirically, extensive experiments on reasoning, safety alignment, and classification benchmarks demonstrate t","title":"ProEval: Proactive Failure Discovery and Efficient Performance Estimation for Generative AI Evaluation","url":"https://arxiv.org/abs/2604.23099","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.23099v2 Announce Type: replace-cross \nAbstract: Evaluating generative AI models is increasingly resource-intensive due to slow inference, expensive raters, and a rapidly growing landscape of models and benchmarks. We propose ProEval, a proactive evaluation framework that leverages transfer learning to efficiently estimate performance and identify failure cases. ProEval employs pre-trained Gaussian Processes (GPs) as surrogates for the performance score function, mapping model inputs to metrics such as the severity of errors or safety violations. By framing performance estimation as Bayesian quadrature (BQ) and failure discovery as superlevel set sampling, we develop uncertainty-aware decision strategies that actively select or synthesize highly informative inputs for testing. Theoretically, we prove that our pre-trained GP-based BQ estimator is unbiased and bounded. Empirically, extensive experiments on reasoning, safety alignment, and classification benchmarks demonstrate t","title":"ProEval: Proactive Failure Discovery and Efficient Performance Estimation for Generative AI Evaluation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-03T04:43:57Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.23099"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ab3247ee8a399187510cb689b8dd45883f18cfcf55b2839f1a3b92e2735f4ef71347fc5cfdcacff930da55ac59c926da95a4ba02a5610b5738d123a09d30d002","signer":"crovia.substrate","subject":{"observed_at":"2026-06-03T04:43:57Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.23099"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"e32c2fd58b5bc90d8571b3dc2bf591cea8ce9c97d2dee61fa66fead3af53a247","leaf_index":209414,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7f77b458b0dff2a8dc25817c0f02d528559abd5110bd93d9b7595ddbfb372e9b","side":"right"},{"sibling":"311ad7c1975a7dc7f6dfa6e6150adf91a8dfbe00426f6216fe60478126f13fbf","side":"left"},{"sibling":"73543b7e064cf9080609451108ed6d861358d70577beeda2560a516c5cc93f61","side":"left"},{"sibling":"9170b1e6483f23a764fb90d1ef5a698d3866d27529919291847d6483d69f9489","side":"right"},{"sibling":"d3a0c352c6a835832575addac4bb4d0aee2bb9f0e1ec97ef852d131d3645de5b","side":"right"},{"sibling":"d559fd51a6779addbcf4873e4b3f57dd068a0a4e618a552735f68c4c42472a78","side":"right"},{"sibling":"11ca635b4b48a5b191a023a3e865a6eb46cdcbad96ba62c238e770638277d969","side":"right"},{"sibling":"bd3f18ac9afb820a8496e447be39bc88ed2f38fc07726a205afbf4d7a97200b4","side":"right"},{"sibling":"5644262858e1dd0ce48f35fced719d068af594f17e93eb1cc769fc5838adcf49","side":"right"},{"sibling":"064aa099decb52d50f639e1542dae2860403f9fca4f561c55f796d74f4af6a83","side":"left"},{"sibling":"2b6b45743f97ac502854e489ac38a3366f8ae7ede58a2728b45daadbf29e9d03","side":"right"},{"sibling":"a81babbd79ea0da9e030dd7f43bffb6519d214317decd50727bea4e78189d970","side":"right"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"8d3baa674a45fd8bc6d3e8d25298f4bec86c72fa8259d576c330d12955c7b4f7","side":"right"},{"sibling":"e32819d1eff909db08066d1703f2db3f091cddac19378b2c0c625ab11e3fdbc0","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":209569,"merkle_root":"c852efc8ad7dfffc196c71380f79e6398bcaf566974cbae7c9950b0f600d5bc8","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260603T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-03T05:37:57Z","sig_algorithm":"ed25519","signature":"1ca417607effc813341721cfdade0a2d2d4ba96a90d361dbc3e864b6991b90abc326ec41d9ba3e7110cc04adb29d7b8d95abea7ff2ab8c591b2f864ed7270c03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_dbf20cb313687e72db75741cdcf2800d697209dc5c6ef20ed40d9de533db16cf"}}