{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_a400b021e79fa254bc51a372498089c81b49c07470e3e80e27eae2e3fe72c0a1","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_a400b021e79fa254bc51a372498089c81b49c07470e3e80e27eae2e3fe72c0a1","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"4acc78886ab5bb2d20c22871beae4d8adf2da42fd535313a25b9546dd4eff444","published":"Tue, 30 Jun 2026 00:00:00 -0400","receipt_hash":"4acc78886ab5bb2d20c22871beae4d8adf2da42fd535313a25b9546dd4eff444","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"4acc78886ab5bb2d20c22871beae4d8adf2da42fd535313a25b9546dd4eff444","observed_at":"2026-06-30T04:43:04.087680Z","parent_run_hash":"74f7ab392cc702044101fe24a76a2fdad11164cd79ce725aad6c446a477e89c5","published":"Tue, 30 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.29088v1 Announce Type: cross \nAbstract: There are various benchmarks to evaluate bugfixing capabilities of Large Language Models. However, most widespread benchmarks do not fully reflect real-world bugfixing practices. They are small, weakening statistical reliability, and the buggy programs are often similar to one another, potentially distorting evaluation results. The range of bug types can also be narrow, failing to capture a representative range of bugs. To address these issues, we introduce MegaBugFix, a large-scale bugfixing benchmark containing 12,629 buggy Python programs synthesized from correct ones by a Large Language Model. Bug injections were generated as diffs representing code changes. Through this approach, we were able to avoid common pitfalls of LLM-based mutation techniques like injecting overly simplistic bugs or failing to modify the input program. We evaluated 13 open-weight models on MegaBugFix and baseline benchmarks, finding consistently lower perfo","title":"Diff-Based Code Corruption using LLMs for Large-Scale Bugfix Benchmarking","url":"https://arxiv.org/abs/2606.29088","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.29088v1 Announce Type: cross \nAbstract: There are various benchmarks to evaluate bugfixing capabilities of Large Language Models. However, most widespread benchmarks do not fully reflect real-world bugfixing practices. They are small, weakening statistical reliability, and the buggy programs are often similar to one another, potentially distorting evaluation results. The range of bug types can also be narrow, failing to capture a representative range of bugs. To address these issues, we introduce MegaBugFix, a large-scale bugfixing benchmark containing 12,629 buggy Python programs synthesized from correct ones by a Large Language Model. Bug injections were generated as diffs representing code changes. Through this approach, we were able to avoid common pitfalls of LLM-based mutation techniques like injecting overly simplistic bugs or failing to modify the input program. We evaluated 13 open-weight models on MegaBugFix and baseline benchmarks, finding consistently lower perfo","title":"Diff-Based Code Corruption using LLMs for Large-Scale Bugfix Benchmarking","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-30T04:43:04Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.29088"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1901ba7108f059ea21e136cb86f932bb6f6ddacec2ae64010ca97a6adcf1e1181aa54b6ca01e0d0c42744a2360cf180a18009649d9cf9065670e787b638f090f","signer":"crovia.substrate","subject":{"observed_at":"2026-06-30T04:43:04Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.29088"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8a51525457f344a963f2ce1ef03c0cccc83f80846f137cd1df37a06f4c4f921f","leaf_index":264858,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"0cca5dd8a82ac337ac2ee100186ea998cd5fcfa4328e066c5e30555f15a2c1f9","side":"right"},{"sibling":"b0ff7cf940ae8f3571aedf7d245dc27369211265b3d2bd6eaa46851bca81c92b","side":"left"},{"sibling":"18685154f93984de48a981bb2e8b330e528c7889b00358b0bdb6ccfb48d3bc8c","side":"right"},{"sibling":"e47cafe1c2f42fd05e6f338dc858064a5eba22c3f8fb239d99b122c121618b4a","side":"left"},{"sibling":"d09e5cd545c7724aceb8f9cd56ac32a1350ebec6e837d965ff60f65e3d135d34","side":"left"},{"sibling":"21aeb247f596e755f92e03c02397140679f193fb3b000cc11078505b42970abc","side":"right"},{"sibling":"07528e32fda4a58ff6ab8ca2268cc78d4ff4ed66bfadde9432f593a2fb4e74e7","side":"right"},{"sibling":"86ba07f5dfcbf22e044b5aa0ecc15cd2f29de9218e54da603c4ffe753c522516","side":"left"},{"sibling":"5767f14b55df4212d25ae2f52b8f3d4abff2555c65ba62935ed6cb0a8e617b43","side":"right"},{"sibling":"1549a8883ab3267f958dc2624919e40c65c82958b8005967ed6a4da1247da0ad","side":"left"},{"sibling":"23994bf0974e5c9c7f63a61b4f0a48b0ca756a4adc34a8f85f878e774c37dfbe","side":"right"},{"sibling":"f9b4bed84fa6990c71ad2887c91bda183001f05f6b648f21d1273045a26b11fd","side":"left"},{"sibling":"173d2dc4b29ee04ea41d6d0ebc334c4bc2d46e7ee4230c94765413f24fb4bc42","side":"right"},{"sibling":"112461f7c0ec411116fb5c6c90fe95cea9d8f188b9fe08afe25a837ac02d0071","side":"right"},{"sibling":"ea9488204352c49db8f7daf05eefcd7628ecf9413830346674801a99d0654a94","side":"right"},{"sibling":"6261c13b9922cb657f10d1e5d36ec15d8771cf8766e36c61dcbffb7bed57e396","side":"right"},{"sibling":"fa19aa3faf287618b820bcfceebb366152ad521dd20ef9f51e977816663e448b","side":"right"},{"sibling":"c32f943406b62d1fc59b7f7e243492174c8e1caba8c8a2705f86c773315736e0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":265374,"merkle_root":"9636001ecab173cb6af10dc7c71eb14585daa62f9c0a6f027046f05633156891","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260630T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-30T05:38:03Z","sig_algorithm":"ed25519","signature":"6d4b8fd9b9da856cbb5fba7540877c6a63fa18a5ec3eaf28dc4d6d1c64921c9c0565f8c95f4b7aec0e7744fd7754844f051baf863db708cad768765b416a7b0c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_a400b021e79fa254bc51a372498089c81b49c07470e3e80e27eae2e3fe72c0a1"}}