{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_7037c2c9580c96418f1077e0509b1dd986f69052efcd127a5cc6a9d5a95de0e2","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_7037c2c9580c96418f1077e0509b1dd986f69052efcd127a5cc6a9d5a95de0e2","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"1b45d620d1f180a773ab57dfa3a07f904801d1a8a56c133df30c8e2bef420cae","published":"Thu, 28 May 2026 00:00:00 -0400","receipt_hash":"1b45d620d1f180a773ab57dfa3a07f904801d1a8a56c133df30c8e2bef420cae","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"1b45d620d1f180a773ab57dfa3a07f904801d1a8a56c133df30c8e2bef420cae","observed_at":"2026-05-28T04:43:38.862500Z","parent_run_hash":"58f8b4a134069e0a15ea3949252489597eb86dd27c9ca3fb15c6fb838ce49ef3","published":"Thu, 28 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.02898v2 Announce Type: replace \nAbstract: Language model benchmarks are pervasive and computationally-efficient proxies for real-world performance. However, many recent works find that benchmarks often fail to predict real utility. Towards bridging this gap, we introduce benchmark alignment, where we use limited amounts of information about model performance to automatically update offline benchmarks, aiming to produce new static benchmarks that predict model pairwise preferences in given test settings. We then propose BenchAlign, the first solution to this problem, which learns preference-aligned weight- ings for benchmark questions using the question-level performance of language models alongside ranked pairs of models that could be collected during deployment, producing new benchmarks that rank previously unseen models according to these preferences. Our experiments show that our aligned benchmarks can accurately rank unseen models according to models of human preferences","title":"Aligning Language Model Benchmarks with Pairwise Preferences","url":"https://arxiv.org/abs/2602.02898","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.02898v2 Announce Type: replace \nAbstract: Language model benchmarks are pervasive and computationally-efficient proxies for real-world performance. However, many recent works find that benchmarks often fail to predict real utility. Towards bridging this gap, we introduce benchmark alignment, where we use limited amounts of information about model performance to automatically update offline benchmarks, aiming to produce new static benchmarks that predict model pairwise preferences in given test settings. We then propose BenchAlign, the first solution to this problem, which learns preference-aligned weight- ings for benchmark questions using the question-level performance of language models alongside ranked pairs of models that could be collected during deployment, producing new benchmarks that rank previously unseen models according to these preferences. Our experiments show that our aligned benchmarks can accurately rank unseen models according to models of human preferences","title":"Aligning Language Model Benchmarks with Pairwise Preferences","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-28T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.02898"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:53a22efcbaf4ed7531f8d8a915192fef49edfba3166791a4dad563282d9fcd18334061f98e5b2020a4be8c1bdf436b2b5c8c80db3d2d1b0a44f99c0e3cb77f07","signer":"crovia.substrate","subject":{"observed_at":"2026-05-28T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.02898"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1ae74a0d291aee5cb478e7c0ae074aa2d0ddb0ae908c137d46c7e453737a6769","leaf_index":155958,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"364e758a3eead9132796fa568fd49ddc6f6c5af9c41d228ce40e80adefdb2d79","side":"right"},{"sibling":"3a4acb9357a569c54a1684f07bcdae3faf89ce746e449646abf8b20ec1561bf8","side":"left"},{"sibling":"833d432006d3f349c56e25961422ae8b13e3f326f8441db9d4688870c2b8c795","side":"left"},{"sibling":"983a25e092fd606d79871389baede32762dee114771e1c3dee4810e79d701a6c","side":"right"},{"sibling":"d814af7416b1d7df7e916bfbcaa45277d0f80784891f22c2d080dd1d468b9d6f","side":"left"},{"sibling":"3a80ec8eb78d732ca7dd30743659c452c4b2481f913f50b77b706df8a28bc68c","side":"left"},{"sibling":"6ca55e77be0491ef320b9c77df7aeb010f9059a95b2e0a2f0b2b6582f1a052b3","side":"right"},{"sibling":"23dd40eb30de320ad1061f73adc18c4482f490a6bd28a6d2175e4ff59f041bbb","side":"right"},{"sibling":"92ebfba9adaa779ba57179e1e0f933c2128036a29e48cb81e4e64272dbbbb5c0","side":"left"},{"sibling":"f292d3278e493ec60902181b8c0bd5c89c0a1168ef928222fe6161287982f7f0","side":"right"},{"sibling":"2209295faf1a5bf51c97c6fd5a839a8181a4ea44f490420381530f35df7d9b2f","side":"right"},{"sibling":"5784576a15214ea9fc3569e6e1cff1ef443c0b1fc0d036028489088af089de27","side":"right"},{"sibling":"311772ec218efcb2da5a337f9e9f042fe1cc0028643adb0a354787e4ea7911b7","side":"right"},{"sibling":"66331bac84ca0f8983eb09fac7eaf95af234f1b82680b793eabff4ee25caac40","side":"left"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"7b6f0bea4291a5e63574dfca9aa0f9756f450c9478c3d07474d39a7ababb51f9","side":"right"},{"sibling":"1d39fe14b21e2ebbfb87e882423b24ee9469eae1e4c77af5b799ac4db9537467","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":156177,"merkle_root":"c5705a0243d16afd8b1ebfd731b7aa304079c442c2a7906493c5bbed374c69ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260528T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-28T05:37:36Z","sig_algorithm":"ed25519","signature":"f087e13febc8bb6a2e0812610de64cebc65be92915518d9c4b230799c3b161839b04c4eb1b02741f938f35545a76ab76555b04c782bdc2f9a44852d171d65909","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_7037c2c9580c96418f1077e0509b1dd986f69052efcd127a5cc6a9d5a95de0e2"}}