{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_f034059b402cf5d495577e00e04faa480a4d2c20f012efa8c33ec9f88529fc6f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_f034059b402cf5d495577e00e04faa480a4d2c20f012efa8c33ec9f88529fc6f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"668c4cfeae04b0e190cb8010ecff76be9f2baa8935d4a2a1de7a163d7c359bf9","published":"Tue, 09 Jun 2026 00:00:00 -0400","receipt_hash":"668c4cfeae04b0e190cb8010ecff76be9f2baa8935d4a2a1de7a163d7c359bf9","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"668c4cfeae04b0e190cb8010ecff76be9f2baa8935d4a2a1de7a163d7c359bf9","observed_at":"2026-06-09T04:43:45.619596Z","parent_run_hash":"f2344865fd128464efd1bacba326b5a7ccea707694b8c5650dd51ae8c46ac8a1","published":"Tue, 09 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2511.19829v3 Announce Type: replace \nAbstract: Prompt optimization has become a central mechanism for eliciting strong performance from LLMs, and recent work has made substantial progress by proposing diverse prompt evaluation metrics and optimization strategies. Despite these advances, prompt evaluation and prompt optimization are often developed in isolation, limiting the extent to which evaluation can effectively inform prompt refinement. In this work, we study prompt optimization as a process guided by performance-relevant evaluation signals. To address the disconnect between evaluation and optimization, we propose an evaluation-instructed prompt optimization approach that explicitly connects prompt evaluation with query-dependent optimization. Our method integrates multiple complementary prompt quality metrics into a performance-reflective evaluation framework and trains an execution-free evaluator that predicts prompt quality directly from text, avoiding repeated model exec","title":"Knowing How to Edit: Reliable Evaluation Signals for Diagnosing and Optimizing Prompts at Query Level","url":"https://arxiv.org/abs/2511.19829","vendor":"arxiv_cs_ai"},"summary":"arXiv:2511.19829v3 Announce Type: replace \nAbstract: Prompt optimization has become a central mechanism for eliciting strong performance from LLMs, and recent work has made substantial progress by proposing diverse prompt evaluation metrics and optimization strategies. Despite these advances, prompt evaluation and prompt optimization are often developed in isolation, limiting the extent to which evaluation can effectively inform prompt refinement. In this work, we study prompt optimization as a process guided by performance-relevant evaluation signals. To address the disconnect between evaluation and optimization, we propose an evaluation-instructed prompt optimization approach that explicitly connects prompt evaluation with query-dependent optimization. Our method integrates multiple complementary prompt quality metrics into a performance-reflective evaluation framework and trains an execution-free evaluator that predicts prompt quality directly from text, avoiding repeated model exec","title":"Knowing How to Edit: Reliable Evaluation Signals for Diagnosing and Optimizing Prompts at Query Level","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-09T04:43:45Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2511.19829"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:49e6010b70c87302f4a3cba7a261d0a6e53dcd03f280cbb702c8d696841a78396ce0c98d154a79ff0542bc62952751d7b33be57b0477331f8f368ec81089b505","signer":"crovia.substrate","subject":{"observed_at":"2026-06-09T04:43:45Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2511.19829"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"c6106837ffeb3c513e3cea88236935eeb5d5457dfebcdadded806073cd4e41e2","leaf_index":224457,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"0f7e6556649893d72e3593b0bc76e9526a5bb0b9ec46df188b1c27cf3cc71c3e","side":"left"},{"sibling":"a8b0a309b7ab667c18616062bc1d015f2308820606f91e3d0f70739eafae9471","side":"right"},{"sibling":"3cca94d278a14dd2d187ce50eb9c14105dcf94fae1a54a7a69da5b786996413d","side":"right"},{"sibling":"5aa8b68fb1f1a2763d3c6e7da3ee25d49f3abac37035607d40d340774039320a","side":"left"},{"sibling":"dcfa109865e99f057cf9154701a897872f8b893b161c05113b0e77cfa34cb785","side":"right"},{"sibling":"cda7f11a1cff7b91dee93d3661fadda29240d7915568285d6aace5cc6c68149c","side":"right"},{"sibling":"6544944fc5c0722251532a5665622a7bb9e1790fe437c54f30cf1b0f097f4768","side":"left"},{"sibling":"0dba1b4aa8c69a39c01dbd9d9884405788a8121b073c56815159112e9ccba4bd","side":"left"},{"sibling":"f6fb234a4e2f067b22329eec05b093a8b38f0411de9434d5a8eb55c2f70f1a2f","side":"right"},{"sibling":"b2df6a4bb3e928f0b447931cc688ae01d2415773a2b07cfed0b1cba689078aed","side":"right"},{"sibling":"b1c9ec856caa0fd46bb47b46f18c59ebcd295d774ca17adb3b46f05d394a6a5d","side":"left"},{"sibling":"24fdc29d461691aedb6fa920758206b5bb43851f477ef7a04c34aaed84b8971b","side":"left"},{"sibling":"036922da4e1e2c46d948f070454bfad299b7406fb00735ea9d8bd1e687f5f445","side":"right"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"87c6b850dfec08ac35a693d9db3a3315250a68adb1cfab9b1015f212b63b15bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":224761,"merkle_root":"e9f7b49b652e869ab97ffba9c5a31356b2d0e3dc5d00bb28944adf737c46b1e7","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260609T103805Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-09T14:15:34Z","sig_algorithm":"ed25519","signature":"8ad8076fb12c8e486ae1d1559a9a7ba8e2ee996a9ad3d8ba7bcdbdbd88ab3a15bcb429707aca6d3e9d8b97e2ba755b3dcc77b1abb6601ccb829842719a6fb30d","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_f034059b402cf5d495577e00e04faa480a4d2c20f012efa8c33ec9f88529fc6f"}}