{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_eef360e0467c9f4fbf53a7f5c2f81f4f08bed724ce6bfaca2b6709b6720c8014","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_eef360e0467c9f4fbf53a7f5c2f81f4f08bed724ce6bfaca2b6709b6720c8014","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e2d424bdbf91b80ed0941478faddee7c4cc337307ecdb20f20248eb511c6ee4d","published":"Wed, 20 May 2026 00:00:00 -0400","receipt_hash":"e2d424bdbf91b80ed0941478faddee7c4cc337307ecdb20f20248eb511c6ee4d","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e2d424bdbf91b80ed0941478faddee7c4cc337307ecdb20f20248eb511c6ee4d","observed_at":"2026-05-20T04:43:44.562035Z","parent_run_hash":"5f904c2c2fecb6b44f2adce8bdc9de914b9a39f7c4fdd7bf086a6c2361a30f8c","published":"Wed, 20 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2601.16823v2 Announce Type: replace-cross \nAbstract: Large Language Models (LLMs) exhibit remarkable capabilities, yet it remains unclear to what extent these reflect sophisticated recall or genuine reasoning ability. We introduce chess as a controlled testbed aimed at disentangling these faculties. Leveraging the game's structure and scalable engine evaluations, we construct a taxonomy of positions varying in density of relevant priors - ranging from common states solvable by memorization to completely novel ones requiring generalization. Crucially, our approach achieves this distinction without requiring explicit knowledge of the models' training data. Applying this taxonomy, we combine a longitudinal analysis of the GPT lineage with a rigorous evaluation of contemporary models, including Claude Opus and Gemini. Our analysis reveals a steep gradient: performance consistently degrades as the density of relevant priors decreases. Notably, for tasks with few relevant priors, base ","title":"Disentangling generalization and memorization in large language models using chess","url":"https://arxiv.org/abs/2601.16823","vendor":"arxiv_cs_ai"},"summary":"arXiv:2601.16823v2 Announce Type: replace-cross \nAbstract: Large Language Models (LLMs) exhibit remarkable capabilities, yet it remains unclear to what extent these reflect sophisticated recall or genuine reasoning ability. We introduce chess as a controlled testbed aimed at disentangling these faculties. Leveraging the game's structure and scalable engine evaluations, we construct a taxonomy of positions varying in density of relevant priors - ranging from common states solvable by memorization to completely novel ones requiring generalization. Crucially, our approach achieves this distinction without requiring explicit knowledge of the models' training data. Applying this taxonomy, we combine a longitudinal analysis of the GPT lineage with a rigorous evaluation of contemporary models, including Claude Opus and Gemini. Our analysis reveals a steep gradient: performance consistently degrades as the density of relevant priors decreases. Notably, for tasks with few relevant priors, base ","title":"Disentangling generalization and memorization in large language models using chess","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-20T04:43:44Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2601.16823"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:4fe74ec29eb5dbb128c6083fefbecbf0e6c86c6dd48493d526dcdb416793de90c11086d78bc6004d1de528f839563f83738a3beeac76b366335ca7790a826e04","signer":"crovia.substrate","subject":{"observed_at":"2026-05-20T04:43:44Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2601.16823"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"baabbeaa4a096ff4adbc74ef73b5b152c932b4919d325c14fa5130f90a00eb9a","leaf_index":145282,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"234df0f537126d82ff91f0e682eb47eaf1d92e9ff30ac68476bfde75e68deacd","side":"right"},{"sibling":"3941e779a979f4d88dd94ae1b76ef695e9c60cfbd4d49f1edf2d64fbd24c531f","side":"left"},{"sibling":"58053852d84f1c519df261e9c6a0bc959314fc2bcfdc6c4836a785c6f2bab2b9","side":"right"},{"sibling":"a3cce5cb88fa8ed2144362fb8190b20b970b698d38bfc66cf480aa38be1d96d2","side":"right"},{"sibling":"5f03ce43876e6319740b03425a5aeaeb109f248a221d0d33fd4eee59673f01e9","side":"right"},{"sibling":"01ed1e09ca55b8947e9e2a49f489627f1a042f0c03eac8fb97fed38389096fa5","side":"right"},{"sibling":"cefd93338369caa43fee7488c87e004962f685c84a7eb2dabfe358e69b38d6f1","side":"right"},{"sibling":"cc6b64c1eb047b460e9fe9beb452cf69fe3eb490dcf6d93d8193dda021cdf8ef","side":"left"},{"sibling":"f073aa7be27ee7e1d9eb0de5f129f7e9bae0c584fb959378c395b65831dc1d1d","side":"left"},{"sibling":"1526885f19d1fadf6955cf519dbc4e62d593a4bba99d741e8c301740a7068233","side":"left"},{"sibling":"e3a7d5c07f161682d61bd453ffc02ecdf87cfeda70f986d6650017f9d2d6b265","side":"left"},{"sibling":"edbc49f08e5b92291934c05c9e6efd270a6b0698d8d2fa474006366027dfe098","side":"right"},{"sibling":"3e4df6e7457cecbf36f350375e72dcab336a3984422e4c406ef809e4e2944e96","side":"left"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"be08fedc4e72a6fb56606f66812fae7317e09690b9acb18385f4ab117a981238","side":"right"},{"sibling":"0534329a7475dc9df51998c83c16892126679dade0fa34182f21e869599386c7","side":"right"},{"sibling":"2d24720928ead0e7670650eb55f558c4f20e4c18df376f47ba72cfa8cf0ed344","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":147301,"merkle_root":"08903d7159c3b38eeeeafc09eab15139ea417f1d94f02f1fbc87296b37db840a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260521T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-21T18:37:33Z","sig_algorithm":"ed25519","signature":"905f2924632dfa2970c8690285f5b5d4a1d891d0e0ef1cbc404ebec2fd937215ac768e16f0a9f28b18977a55ae0bfd226db6833ae7ef588729054117d2da7303","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_eef360e0467c9f4fbf53a7f5c2f81f4f08bed724ce6bfaca2b6709b6720c8014"}}