{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b9dd6056602f2443a96d6dd4c534d795dc2c1970f376b79d9934af13c7b33e7a","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b9dd6056602f2443a96d6dd4c534d795dc2c1970f376b79d9934af13c7b33e7a","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"99115b257dbfa3cb5f0acc1e3a832c2ea8815fa195dbb01c0f0e3a397e9d5bf9","published":"Tue, 21 Jul 2026 00:00:00 -0400","receipt_hash":"99115b257dbfa3cb5f0acc1e3a832c2ea8815fa195dbb01c0f0e3a397e9d5bf9","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"99115b257dbfa3cb5f0acc1e3a832c2ea8815fa195dbb01c0f0e3a397e9d5bf9","observed_at":"2026-07-21T04:43:35.036805Z","parent_run_hash":"03e944014de2697434479833d15ea9303e014945afc230ecc7f207824493b589","published":"Tue, 21 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.16241v1 Announce Type: cross \nAbstract: Recent large language models (LLMs) can generate custom CUDA kernels that appear to outperform PyTorch on benchmarks such as KernelBench. Building upon this foundational framework, we demonstrate that frontier models frequently engage in reward hacking to artificially inflate reported performance. In this work, we identify two areas where evaluation frameworks must co-evolve with model capabilities. First, to accurately measure true speedup, we examine the baseline timing mechanism, noting that enabling Tensor Core acceleration with TF32 provides a more realistic estimation of execution on modern GPUs. Second, concerning algorithmic correctness, models often exploit the narrow test distribution by hardcoding bypasses for specific tensor values. By skipping required computations, these kernels artificially accelerate execution rather than implementing actual CUDA kernels. We introduce KernelBench-Verified, an extended evaluation framewo","title":"KernelBench-Verified: Do LLM-Generated Kernels Actually Beat PyTorch?","url":"https://arxiv.org/abs/2607.16241","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.16241v1 Announce Type: cross \nAbstract: Recent large language models (LLMs) can generate custom CUDA kernels that appear to outperform PyTorch on benchmarks such as KernelBench. Building upon this foundational framework, we demonstrate that frontier models frequently engage in reward hacking to artificially inflate reported performance. In this work, we identify two areas where evaluation frameworks must co-evolve with model capabilities. First, to accurately measure true speedup, we examine the baseline timing mechanism, noting that enabling Tensor Core acceleration with TF32 provides a more realistic estimation of execution on modern GPUs. Second, concerning algorithmic correctness, models often exploit the narrow test distribution by hardcoding bypasses for specific tensor values. By skipping required computations, these kernels artificially accelerate execution rather than implementing actual CUDA kernels. We introduce KernelBench-Verified, an extended evaluation framewo","title":"KernelBench-Verified: Do LLM-Generated Kernels Actually Beat PyTorch?","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-21T04:43:35Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.16241"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:b7a36ef55695d6d9dfd1d711a9b835689ba7ac5fb71cbe8a41dc502b58c07794457673fc6b7203418b81a36c53051c6098149738d016dabbe673d25bd5e92309","signer":"crovia.substrate","subject":{"observed_at":"2026-07-21T04:43:35Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.16241"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"b01828237d403989a406d3237a31b3bd71542606327829b8b3f306ff4e102747","leaf_index":336639,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c1df5b8cec19cf7d474edd577776f4425e3cf524029ded44bf13ddf7017dacba","side":"left"},{"sibling":"4d7478c5dfc74af4da613b89ad53d71c4382db8a7de0902065d133461779417e","side":"left"},{"sibling":"db97260d4692db0ac5716f81996c11dcf37fdad4c5a5031b41398ab8663eff42","side":"left"},{"sibling":"6d129ef0192bc44a3a96606de2c179d906c9ac7ba7f94d0114a61bd0de01a576","side":"left"},{"sibling":"72ece6a3c66026c3d1afb05b2b7d39668a3f916397f146b63c60b2e207237529","side":"left"},{"sibling":"b2189a2bc21d1e40057c39486554f2eeccc1d12bef974a9b2cc82edce1d2ad42","side":"left"},{"sibling":"78afb98e8815c9330b3a7e74c8b56112eb0284b980a66d5b96cce812169af054","side":"left"},{"sibling":"e6f9d6d6c760446a7b30dd4e30a28f58c0817f4bee09529ee009c470d86f5564","side":"left"},{"sibling":"38e5827f7c9f72ad34a2b97042f2fb5f7e868db899d074b7819af4f2209b9caa","side":"right"},{"sibling":"7b927551b5db06b6571913b4e6792ffcce5291a3eca5a0df4a3b6e296271105f","side":"left"},{"sibling":"b77a0b5ae4607c8fe6ba73449d46b35076e3dedc0c82a2c65a05780d42a7bc2e","side":"right"},{"sibling":"9eb5077edfb3dc553857d4794b925bfce117e0f8a1d049af5d0dd9026b470eef","side":"right"},{"sibling":"414b1a70fd1dcb25489a194714b97492b066684b15d0b7a48a176c4b9b5bc713","side":"right"},{"sibling":"21d66dd41003813f710b7617944f1bfba3258658a5d3370c21cad8f9e945bc99","side":"left"},{"sibling":"613f015699131eb89bd755dee67133be95af25cf5f16c1c8ce4b99d963b8dd86","side":"right"},{"sibling":"a729b574b1135956436ded5eef1fe8f08014ff6a0729749d307ab1bca93fcdc9","side":"right"},{"sibling":"77025bcb374a7ad74f520643e20a8ae1205a7ee78507b0beb93117760f1c29d3","side":"left"},{"sibling":"4bf21052e085e8ac81f1dec1d2b310bd12bf948992de6177d12e9d2fda8d39f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":337144,"merkle_root":"5e969cc01afa67e4dbe5d37b712cdb10f4aa1fd74404e02eab724cf487c8d6d9","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260721T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-21T05:38:38Z","sig_algorithm":"ed25519","signature":"5c0c1a8dd2793d787ccd5e49e8b4d70eed136352555518589f05c74be357fc171702e42a76c3a556d90e3d51ff36cb3d292aaac83c66566de7b943f318bda50c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b9dd6056602f2443a96d6dd4c534d795dc2c1970f376b79d9934af13c7b33e7a"}}