{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c879508e16d9a87572df8dd7d4e1485fc83607209fdf3c5a3d45825ae1bc926e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c879508e16d9a87572df8dd7d4e1485fc83607209fdf3c5a3d45825ae1bc926e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"17abb0ff0f7a4aef6a6ba10731f717009fe12dc0a704f8f96364edddbc70bcab","published":"Tue, 07 Jul 2026 00:00:00 -0400","receipt_hash":"17abb0ff0f7a4aef6a6ba10731f717009fe12dc0a704f8f96364edddbc70bcab","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"17abb0ff0f7a4aef6a6ba10731f717009fe12dc0a704f8f96364edddbc70bcab","observed_at":"2026-07-07T04:43:08.294902Z","parent_run_hash":"fc40a96e5d33ecc82922806c3ad18de4725d7af03964570396c8af4e48fb5bc1","published":"Tue, 07 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2501.10711v5 Announce Type: replace-cross \nAbstract: Code-related benchmarks play a critical role in evaluating large language models (LLMs), yet their quality fundamentally shapes how the community interprets model capabilities. In the past few years, awareness of benchmark quality has grown. Yet, after a decade-scale (2014-2025) survey over 672 code benchmarks, we observed a lag between growing awareness and actual practice. For example, in 2025 alone, the number of benchmarks that ignore code coverage when providing test cases nearly matches the total count accumulated across the previous ten years. In response, we take a clear position: Code benchmarks must prioritize rigor in benchmark construction, reliability in evaluation, and reproducibility in release. To operationalize this position, we introduce a code benchmark guideline HOW2BENCH with 55 checklists. Finally, our further human study also exposed that the current issues not only stem from the significant effort requir","title":"Code Benchmarks Should Prioritize Rigor, Reliability, and Reproducibility","url":"https://arxiv.org/abs/2501.10711","vendor":"arxiv_cs_ai"},"summary":"arXiv:2501.10711v5 Announce Type: replace-cross \nAbstract: Code-related benchmarks play a critical role in evaluating large language models (LLMs), yet their quality fundamentally shapes how the community interprets model capabilities. In the past few years, awareness of benchmark quality has grown. Yet, after a decade-scale (2014-2025) survey over 672 code benchmarks, we observed a lag between growing awareness and actual practice. For example, in 2025 alone, the number of benchmarks that ignore code coverage when providing test cases nearly matches the total count accumulated across the previous ten years. In response, we take a clear position: Code benchmarks must prioritize rigor in benchmark construction, reliability in evaluation, and reproducibility in release. To operationalize this position, we introduce a code benchmark guideline HOW2BENCH with 55 checklists. Finally, our further human study also exposed that the current issues not only stem from the significant effort requir","title":"Code Benchmarks Should Prioritize Rigor, Reliability, and Reproducibility","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-07T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2501.10711"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:28a41aadcf7cf46504fd4d51ebfcdaf7308c498df4ceb04f35e760426b3f2892ced733572d5f43d1a190d3dca64f3d19ac8a1ad2f3c627046fd70ea21a27180f","signer":"crovia.substrate","subject":{"observed_at":"2026-07-07T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2501.10711"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"039069f6d00663b25b4e1a33f380449393f6286260e1adfab581004c455b46f7","leaf_index":289260,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1b39e4357bbf57877b809894c727e64845a3619b4e9bef43ac83c074b5659489","side":"right"},{"sibling":"110ecdc8d03a42331e1663c9e286f3eec50803b76e0e22cf1607f6455f514059","side":"right"},{"sibling":"eb017921124311cc9192ef5f8233d724dccb6b00817ccaefea01ef510eba9dc6","side":"left"},{"sibling":"f400cdb39d6bd18b4672cc330efb4bbbeec02bce0e1c38e511d133de02b59f3e","side":"left"},{"sibling":"152c1d361729e7313405afb8d577bded66f5f064bbce04c77efedcf619f83a5c","side":"right"},{"sibling":"23e63d35a91a833c6a3b8ee481a65027698d2b56a1b781e34c0c96d6080ba3f9","side":"left"},{"sibling":"04448123cdbf37b9cf59136f7e56f8977c859960e09c32bce71bc2ee499ed1ba","side":"left"},{"sibling":"a6e115fb6d42f8f126d042702270c371c7df6161120ef9b37366edddc165bd28","side":"left"},{"sibling":"35d3e8088ee93171aa479055dcd001cfb6d23925114ddc0908531e54298b0d29","side":"left"},{"sibling":"841129c21a7583176cdc7de281cadfe0e00d04673461e199760cd5128d8cc2d5","side":"right"},{"sibling":"19d6dfd29bc47f35fa02e8fe765277ba9cc3e6da5072309f24ebaac5b5f295e3","side":"right"},{"sibling":"8e0ad7889eb2d4b40e5b6c3d8e2eb19d4e202374983f468aa76321823de07a9f","side":"left"},{"sibling":"aae716235efcb893a1f219dbcd5095070d08a497769fc6d50c14976aa26d5750","side":"right"},{"sibling":"a75ab4319e241beeddb1b3f5705febe0422937926c3479923ccfb0b0082fa4e3","side":"left"},{"sibling":"bd04fa605f883bfb2b81510d045b1e85e555a03da3be083619f61384dfe40ff8","side":"left"},{"sibling":"1b72ad8d12164fdf329e7871711be99d8569d140b21f94056e6962da21da9ce1","side":"right"},{"sibling":"5f5109c2bfdcc7a7e70554bba25862e2d7ce86b6b0cd48a72eb66d2eb735f321","side":"right"},{"sibling":"05fd8a05dddb2e7f72bbb5b290ca55c378f1aed709f132277908d9a5f30eb605","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":289613,"merkle_root":"dc428b9d9ba248d4f93f63147bf7c700bf5be7f500cec6c3507b9df6e9401601","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260707T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-07T05:38:15Z","sig_algorithm":"ed25519","signature":"c468b0e183383ab71992be40bda451093e6cd8cd8efb0d26f68e135a804b287c209d12a0f4fdd95c69c835c04b78df8cb1903dee1f53d4730b36f5332a29fe05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c879508e16d9a87572df8dd7d4e1485fc83607209fdf3c5a3d45825ae1bc926e"}}