{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_9f539f0ea7003374b183711ef61672a7826af8d4af983f443e8813b930a414ce","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_9f539f0ea7003374b183711ef61672a7826af8d4af983f443e8813b930a414ce","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"1f3aa9d0f20914fbb45e095b5200d54b6a34bdf2c35a1488b734e82bc1b4197f","published":"Fri, 17 Jul 2026 00:00:00 -0400","receipt_hash":"1f3aa9d0f20914fbb45e095b5200d54b6a34bdf2c35a1488b734e82bc1b4197f","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"1f3aa9d0f20914fbb45e095b5200d54b6a34bdf2c35a1488b734e82bc1b4197f","observed_at":"2026-07-17T04:43:38.280949Z","parent_run_hash":"113193614a8af99887180226d4e28a8b71d957da5fe3694f0e7a56807c145504","published":"Fri, 17 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.14109v1 Announce Type: cross \nAbstract: Probing the capabilities of Large Language Models (LLMs) and building robust solutions for Multiple-Choice Question Answering (MCQA) remain central challenges in natural language understanding. Furthermore, the rapid proliferation of LLMs has created the implicit assumption that more sophisticated prompting techniques yield better performance. Several studies claim better performance with more sophisticated prompting techniques, but do not provide a comprehensive evaluation. We address this gap through a comprehensive empirical study of 8 prompting techniques across 10 multiple-choice question answering (MCQA) datasets, encompassing 27 model configurations and roughly 4,300 unique questions evaluated more than 430,000 times. Our findings reveal a striking paradox that baseline prompting consistently outperforms complex reasoning techniques on various benchmarks. Only minimal expert and inductive role framing (CoT-Expert and CoT-Inducti","title":"Simplicity Paradox: Debunking myths about prompting and datasets for LLM evaluation","url":"https://arxiv.org/abs/2607.14109","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.14109v1 Announce Type: cross \nAbstract: Probing the capabilities of Large Language Models (LLMs) and building robust solutions for Multiple-Choice Question Answering (MCQA) remain central challenges in natural language understanding. Furthermore, the rapid proliferation of LLMs has created the implicit assumption that more sophisticated prompting techniques yield better performance. Several studies claim better performance with more sophisticated prompting techniques, but do not provide a comprehensive evaluation. We address this gap through a comprehensive empirical study of 8 prompting techniques across 10 multiple-choice question answering (MCQA) datasets, encompassing 27 model configurations and roughly 4,300 unique questions evaluated more than 430,000 times. Our findings reveal a striking paradox that baseline prompting consistently outperforms complex reasoning techniques on various benchmarks. Only minimal expert and inductive role framing (CoT-Expert and CoT-Inducti","title":"Simplicity Paradox: Debunking myths about prompting and datasets for LLM evaluation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-17T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.14109"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:48d3c285e195c3be3061d75183a75855005806add66395895452aae5124ff2378f4c21f91f9dac27de1eb46ca9ad024f72b74921212e517057e124a3c888fd0a","signer":"crovia.substrate","subject":{"observed_at":"2026-07-17T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.14109"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"37303a45ce13c69441641dc72aceed825f1292962237919d2bbbeca51c8e5f84","leaf_index":323064,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"38aec143d7d4280dfca631cae2838137c1bd41b5852621224a0c856e199d2d80","side":"right"},{"sibling":"7e04873fd766971f7caaa905851f74846dac74f06662ad77814472de8f37bb48","side":"right"},{"sibling":"524725ea8df3ba72edfd773b6bb336b160a8e40ee6970a1cc9a9d050c637fdb8","side":"right"},{"sibling":"7477fb3dc300b9a20678a49866a886214f688112a10d9a0ba9a23dcd0668c457","side":"left"},{"sibling":"1f5263709dd2d0784570142a5bb0c2ded78cab588dab919bd3e872f4bc8508b1","side":"left"},{"sibling":"f4935e2c2f77b7da9433b98e135fe5e3ce776c72aa0f30e9f3f0db9633de8e85","side":"left"},{"sibling":"fa46cdb7cb881d614aa37b466168cf41201fba62306710750ffddfb495240293","side":"left"},{"sibling":"5944e6d0988d989a4ae4330320fe2633624339ee8db50eaa0d3fe8981ac982cf","side":"left"},{"sibling":"e90a96bd60cc2aad32b2969366989aa5d4e0c71aeb240044c7778cca7188dc3f","side":"left"},{"sibling":"d7e66b08e4f129ea3ff266d25de4f974fe1f0d56e2fae2a6d3ae1e1565007da9","side":"right"},{"sibling":"05a09763743cdc09fc45cf454e4e3ea4a0d1cd74f9c8162b2a57e2c873160908","side":"left"},{"sibling":"de3120ef2488b8a791a686b47257da4e612256abdfcdda7519265e7edd47d041","side":"left"},{"sibling":"e86f56a4883492da5b5e7b0201324c52946e865e69b99ebb532f41fe3c658ee4","side":"right"},{"sibling":"34d85f6ad6cc7dfa79d90e2b9ff99a561bcdc75b0301bbbd3e83861f54535c1e","side":"left"},{"sibling":"f302542c38ba7c3aab7c9280dd60259ecec777dca6e6f71b6f0729b0b8791b72","side":"left"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"9b11714124b9b951ff9450b0ee9d625a0b70b6da2cf388bd3df1475eec0b17ba","side":"right"},{"sibling":"a4523a9014d45df43e006e9210a73428c380d771f2c650a1b986910759b0cdf7","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":323382,"merkle_root":"2f4d32419c80a9600aba5a480fc3fb7012ec0a695c91a1b055048e78760b65ca","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260717T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-17T05:38:31Z","sig_algorithm":"ed25519","signature":"495308c7bf004117807331d3f71d0b079f6bd7ed7737faa58c773b1b3e80ee84928d2d8519cc7d2b809506501aada1be6546f72c7ecda3dad5445cbb44502209","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_9f539f0ea7003374b183711ef61672a7826af8d4af983f443e8813b930a414ce"}}