{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_4375a2d66e05a319cd9821a22ff428a1f3cfb4ff1ffc7ab583d7d72129237b7b","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_4375a2d66e05a319cd9821a22ff428a1f3cfb4ff1ffc7ab583d7d72129237b7b","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"0e349f903f822d01abc39b26ba5019456a9ef825801d841943a351b5f381479a","published":"Tue, 19 May 2026 00:00:00 -0400","receipt_hash":"0e349f903f822d01abc39b26ba5019456a9ef825801d841943a351b5f381479a","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"0e349f903f822d01abc39b26ba5019456a9ef825801d841943a351b5f381479a","observed_at":"2026-05-19T04:43:36.782648Z","parent_run_hash":"fefa4c726316a95c5dda9fc1ca07a38a811cf7ffa9825b2f09b365abacd9b32d","published":"Tue, 19 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.17831v2 Announce Type: replace \nAbstract: Evaluating the reasoning capabilities of Large Language Models is increasingly challenging as models improve. Human curation of hard questions is highly expensive, especially in recent benchmarks using PhD-level domain knowledge to challenge the most capable models. Even then, there is always a concern about whether these questions test genuine reasoning or if similar problems have been seen during training. Here, we take inspiration from 16th-century mathematical duels to design The Token Games (TTG): an evaluation framework where models challenge each other by creating their own puzzles. We leverage the format of Programming Puzzles - given a function that returns a boolean, find inputs that make it return True - to flexibly represent problems and enable verifying solutions. Using results from pairwise duels, we then compute Elo ratings, allowing us to compare models relative to each other. We evaluate 10 frontier models on TTG, an","title":"The Token Games: Evaluating Language Model Reasoning with Puzzle Duels","url":"https://arxiv.org/abs/2602.17831","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.17831v2 Announce Type: replace \nAbstract: Evaluating the reasoning capabilities of Large Language Models is increasingly challenging as models improve. Human curation of hard questions is highly expensive, especially in recent benchmarks using PhD-level domain knowledge to challenge the most capable models. Even then, there is always a concern about whether these questions test genuine reasoning or if similar problems have been seen during training. Here, we take inspiration from 16th-century mathematical duels to design The Token Games (TTG): an evaluation framework where models challenge each other by creating their own puzzles. We leverage the format of Programming Puzzles - given a function that returns a boolean, find inputs that make it return True - to flexibly represent problems and enable verifying solutions. Using results from pairwise duels, we then compute Elo ratings, allowing us to compare models relative to each other. We evaluate 10 frontier models on TTG, an","title":"The Token Games: Evaluating Language Model Reasoning with Puzzle Duels","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-19T04:43:36Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.17831"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:6a31c627321042ba5d2f40113fd3ed0922e53a9a003834630ca7960e866fe0fef2db89418089eddcb974c09dc02a53d2927a53390dc56f1f8f85cf99357dc400","signer":"crovia.substrate","subject":{"observed_at":"2026-05-19T04:43:36Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.17831"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1952045acde7b9b78ecc1f9ee3e661fbe997c674296edb11f596c4d1583ba387","leaf_index":142964,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f8e1761eb58dafdd9e84feb3c8934036c7035724af4e5edae7eb806df4274bb6","side":"right"},{"sibling":"7b69dfa0abee2f80c8718afafebc54e7e144c88baffe7998fa399953c78f6421","side":"right"},{"sibling":"1a3694c2942fee4e5557c49f19c29fb5cafa48bbb3a5160f272d632308c93d0c","side":"left"},{"sibling":"d106015512309d8651244c72aa88526731af31c7b4a42119312f1f58d937bd7e","side":"right"},{"sibling":"ddaacd9c90cab459721d12246ec0ad294daa10209c41ad656cb6729d96fa1fac","side":"left"},{"sibling":"3f9a5345017ea65e5d058e222a647991f120e8ad55377a6ea03ee8db431d977d","side":"left"},{"sibling":"18920e627055779d158c7c223ef853d7d69b12953cbd8e0326b3401e85b3e0a0","side":"left"},{"sibling":"1d2edc28d27d4c312510482f7321ca401bdfe21a8e83916a1ba677f0ea4fcd2b","side":"right"},{"sibling":"78a19962a2f444f055541381645626e3d4e1c8c7c2811bf54eeb89100f89f5b2","side":"right"},{"sibling":"db97141c585f6a1e6bebe92b3ea300ea0f38a2321ca286d85850b11b2dd162a6","side":"left"},{"sibling":"202f1bead178ef3785968d50d3d188264a95192a077654c331612e04a34cbfbe","side":"left"},{"sibling":"72249c8c8b068386e35d16f4bd0bbeb9ba820ca217ef0f0d28396c9fe493f5f0","side":"left"},{"sibling":"ea64599340f7ffdf17ad0cbc1d9401ef8870a347e3847bdc106d06b1673df09c","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"4db1f363729507e27a60851cf6ed334d7b9acdef194ed7d419aba4d2bd367a4a","side":"right"},{"sibling":"a86ee18c45e7fcc408b6007eaece05aa75b2d9ae30252e9e878462b4dffbef7b","side":"right"},{"sibling":"1d18e7663d43ccff0122ecc7ee12645bb16afb607b218e81b1ea2408f863cb78","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":143302,"merkle_root":"999156d40a7c61d9ddd52b7338f3cbda3e68f53bace070c7b616ea194e23b123","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260519T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-19T05:37:30Z","sig_algorithm":"ed25519","signature":"b1a252cc66ff32bed1d10dd88a6b2a200e3856d3dbcfcc4ee55e02e00f3d548e854ed9c544704b222bd5d315492c4a935ba2d90d727c585a67899b0ad602fc05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_4375a2d66e05a319cd9821a22ff428a1f3cfb4ff1ffc7ab583d7d72129237b7b"}}