{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_21ea28df4296e26d04e0b6ef8b57444cd937ac6e0de55b6d699acbe98a9e402c","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_21ea28df4296e26d04e0b6ef8b57444cd937ac6e0de55b6d699acbe98a9e402c","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"9b49dfb27db1598835921c4648cbf783432172e11e0ea2642cb305ec80b85e20","published":"Mon, 18 May 2026 00:00:00 -0400","receipt_hash":"9b49dfb27db1598835921c4648cbf783432172e11e0ea2642cb305ec80b85e20","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"9b49dfb27db1598835921c4648cbf783432172e11e0ea2642cb305ec80b85e20","observed_at":"2026-05-18T04:43:11.219741Z","parent_run_hash":"a8aad7414ebb6b75c726f09cd673410576a7f87e191fbdb9ddac99e9b2b95a05","published":"Mon, 18 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2504.08300v5 Announce Type: replace-cross \nAbstract: Benchmark-based evaluation, e.g., multiple-choice questions (MCQs) and open-ended questions (OEQs), is widely used for evaluating Large Language Models (LLMs), yet their reliability is undermined by benchmark contamination. When pre-exposed to the testing benchmark during training, less capable LLMs have been found to achieve inflated performance, thereby yielding erroneous results in LLM evaluation. In this study, we reframe contamination as an inherent aspect of learning and seek to disentangle and expose genuine capability acquisition from superficial memorization in LLM evaluation. Following this, firstly, by analyzing model performance under different memorization conditions of MCQs, we uncover a counterintuitive trend: LLMs perform worse on memorized benchmarks than on non-memorized ones, indicating the coexistence of two learning phenomena, i.e., rote memorization and genuine capability learning. To disentangle them, we ","title":"Large Language Models Could Be Rote Learners","url":"https://arxiv.org/abs/2504.08300","vendor":"arxiv_cs_ai"},"summary":"arXiv:2504.08300v5 Announce Type: replace-cross \nAbstract: Benchmark-based evaluation, e.g., multiple-choice questions (MCQs) and open-ended questions (OEQs), is widely used for evaluating Large Language Models (LLMs), yet their reliability is undermined by benchmark contamination. When pre-exposed to the testing benchmark during training, less capable LLMs have been found to achieve inflated performance, thereby yielding erroneous results in LLM evaluation. In this study, we reframe contamination as an inherent aspect of learning and seek to disentangle and expose genuine capability acquisition from superficial memorization in LLM evaluation. Following this, firstly, by analyzing model performance under different memorization conditions of MCQs, we uncover a counterintuitive trend: LLMs perform worse on memorized benchmarks than on non-memorized ones, indicating the coexistence of two learning phenomena, i.e., rote memorization and genuine capability learning. To disentangle them, we ","title":"Large Language Models Could Be Rote Learners","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-18T04:43:11Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2504.08300"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ff9034f3de0b65793d26f5a713caf42df01fd9d1a0c1c83265a6a5799af6a2339dc65ba8182ac4e5b883fc628551e4fa6cd6ca8ef2e4d0257add354fe0178e08","signer":"crovia.substrate","subject":{"observed_at":"2026-05-18T04:43:11Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2504.08300"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"fac1654b3659795468f87361aadd00aa85c784c47597cb6f3dd04013d6317be4","leaf_index":140743,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"8f859c44f2befd1f02b8059e10f502e4d4a394c4def4cac3b10bff02cbe4f977","side":"left"},{"sibling":"68ac8742c0e37435359a5f6a522f686fd9680111d6f4fd1ed203b6622489e736","side":"left"},{"sibling":"5b9d53c621e548abc1aa021987291e2cd1a031e2d43db0e2beec7a3919c0e542","side":"left"},{"sibling":"9429b62b9b2be249184d0032c6b3f17e6c9002f58f979e84ffe9b595bb673732","side":"right"},{"sibling":"acbba89483bfd7b7a80d9908a23d691e18a1ad01faef19f498d75267aebfd72c","side":"right"},{"sibling":"195278f3c49cd4ed79b69f218ba282c8c3b02923aea9de624cbd7796075a1b5d","side":"right"},{"sibling":"518d2c79e4bbff1e18ea05bbb3663a3fb58b3b1682a658f34e706ea23e091089","side":"left"},{"sibling":"30c60828b6b0ade79197e585b88c06ccf4f348c1e62fbf6710d89ae2c2e9dcfb","side":"left"},{"sibling":"6bedf73520cf3dd8758d8bdedf3be245de9aea97abd42934aae25539176ae1b2","side":"left"},{"sibling":"07abc3bad689e74e6304772503dc9372a118e6f66883b8e88c43414efddac063","side":"right"},{"sibling":"28b78fb112bcf26b6801664db97eb8f52a9bccbf0a7ae6766e11845d443692df","side":"left"},{"sibling":"68d0a4634c1460a19c92edd9480df3aa733b814463e7420d1e14471bf61b2f83","side":"right"},{"sibling":"8af64f275b862349aa3bbb9d5cd7fa9a7fdd5620af3bf1b36b2a4519b0b53bdf","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"b98c2afadb358e5387e88f19588f8343a81b488d9b44a6f7e57a032db3a1b030","side":"right"},{"sibling":"11b0c1591747f09f7c8971a6caa19befcd81317ca9dfd417b143234df4e10c79","side":"right"},{"sibling":"87206f3bcc342797c990d87f7235c01f78d32ca59cfaf8ad18d71afc879ba477","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":140892,"merkle_root":"6cca56ead155990456b8a014cc50bddbe710f409b26e3d1bfa6fb12b0bfcf6bf","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260518T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-18T05:37:30Z","sig_algorithm":"ed25519","signature":"1e1135f7595f79b14fb11f5fa81a2e17ad31b11b44b427a5e40a7d511cd86447daf492babd368ab571cf26404c8c74c450d460130fca4b064eb2760367489a0f","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_21ea28df4296e26d04e0b6ef8b57444cd937ac6e0de55b6d699acbe98a9e402c"}}