{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_0da34647321f103b416d2d08914c26908431147250c63bfad16adcc2b70dec96","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_0da34647321f103b416d2d08914c26908431147250c63bfad16adcc2b70dec96","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"be9869d4be963538fd7fe35cfe072b4caa4290613531bb600382747b3170b8a4","published":"Thu, 11 Jun 2026 00:00:00 -0400","receipt_hash":"be9869d4be963538fd7fe35cfe072b4caa4290613531bb600382747b3170b8a4","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"be9869d4be963538fd7fe35cfe072b4caa4290613531bb600382747b3170b8a4","observed_at":"2026-06-11T04:43:37.662146Z","parent_run_hash":"5267801b61ae0d882196b5f37208a9a1633905a64ca7d933f1fa5075cd861491","published":"Thu, 11 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.12117v1 Announce Type: cross \nAbstract: Benchmark scores often misrepresent a large language model's (LLM's) knowledge, because they rely, e.g., on the model's ability to follow specific formatting requirements. This especially penalizes base models that may know the correct answers but lack the ability -- typically introduced in post-training -- to structure them as instructed. To overcome this, we propose soft-prompt tuning, an efficient, fair, and architecture-agnostic model evaluation. By optimizing only 10 soft-prompt vectors (roughly 0.0006% parameters for a 7B model) over a short tuning period, we adapt models to specific benchmark formats, closing gaps in format-following and ensuring that underlying knowledge is accurately reflected in benchmark scores. This allows one to fairly compare different base models -- trained with various pre-training recipes -- on benchmarks without the need for full post-training. We evaluated soft-prompt tuning across 7 models and 7 dat","title":"Soft-Prompt Tuning for Fair and Efficient LLM Benchmark Evaluation","url":"https://arxiv.org/abs/2606.12117","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.12117v1 Announce Type: cross \nAbstract: Benchmark scores often misrepresent a large language model's (LLM's) knowledge, because they rely, e.g., on the model's ability to follow specific formatting requirements. This especially penalizes base models that may know the correct answers but lack the ability -- typically introduced in post-training -- to structure them as instructed. To overcome this, we propose soft-prompt tuning, an efficient, fair, and architecture-agnostic model evaluation. By optimizing only 10 soft-prompt vectors (roughly 0.0006% parameters for a 7B model) over a short tuning period, we adapt models to specific benchmark formats, closing gaps in format-following and ensuring that underlying knowledge is accurately reflected in benchmark scores. This allows one to fairly compare different base models -- trained with various pre-training recipes -- on benchmarks without the need for full post-training. We evaluated soft-prompt tuning across 7 models and 7 dat","title":"Soft-Prompt Tuning for Fair and Efficient LLM Benchmark Evaluation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-11T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.12117"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:16b30c12bf83056cca95974cf1c3405c1dfe861200b8d01681e80db7928d1125a2db68c609b6b32948eb978e485bfdfb58f4f63e6cd4342d14802117699f5509","signer":"crovia.substrate","subject":{"observed_at":"2026-06-11T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.12117"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"973b3a4308c9b4a54ba8ffbb7331a6c51b83d93e1f9818ecfc8f40fad7412450","leaf_index":227486,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"366ca24fb59d4cd21b2e976736ad8ef546ad9fdfb302fc557d9a9c0beb020391","side":"right"},{"sibling":"f9cbd488c940579a610536eeb082ea973971473912f61d692512d985c55bedc3","side":"left"},{"sibling":"ece96c0767b6666136b92cdd85cd8c5e67608722c628194e6c55c5a8692f9058","side":"left"},{"sibling":"b81e94ccf311e9a4bbf0d2d70bbec568be4d7fa5c5d904862b888c10eac526a3","side":"left"},{"sibling":"7e9717ea895cd99e4039b58eeaf845b2168a1b1111348e2afeac9a228bfb863a","side":"left"},{"sibling":"4675cdd5067e9878d980d32ed6d49a2f4c959cec55d570d1bb2ac2247566333a","side":"right"},{"sibling":"be2a37bcb238a826acf45455e518d42d73f440ba96d52faa79e4bc8b1793ea0f","side":"right"},{"sibling":"e5516844190de8c773a2f33a88fdd935790a83aa93beef323f043368177406bc","side":"left"},{"sibling":"2cfac7f042209c8533c6031bddc4a83bc156e0595f1b28efefbda208904f338a","side":"right"},{"sibling":"04e399458c5b36988cae0bf1c6dbe1b01349003b15cb5aa43f95c55acffe4ec3","side":"right"},{"sibling":"1383228337d54218bd8e5563aebb0b0dfe15c5269e3d5138e64c261d6130a88b","side":"right"},{"sibling":"57cb49c192550231071a0bf53a0821da2f79c585ec6c8d0fc76cebd62ccd78b2","side":"left"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_0da34647321f103b416d2d08914c26908431147250c63bfad16adcc2b70dec96"}}