{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1a4649430b887c1bbafde15f4738a0344e04b5c77445297ea4cf2dea8e8cfa60","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1a4649430b887c1bbafde15f4738a0344e04b5c77445297ea4cf2dea8e8cfa60","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"de0172d9dd3ccbab01d1e05514358c9c8db294e67ecd28f2abad492fff20bab8","published":"Thu, 04 Jun 2026 00:00:00 -0400","receipt_hash":"de0172d9dd3ccbab01d1e05514358c9c8db294e67ecd28f2abad492fff20bab8","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"de0172d9dd3ccbab01d1e05514358c9c8db294e67ecd28f2abad492fff20bab8","observed_at":"2026-06-04T04:43:08.243501Z","parent_run_hash":"298818240313a3c9ce3ace3750dbf55845013dc3bc59aeced6331eccefdb61ac","published":"Thu, 04 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2511.05722v3 Announce Type: replace-cross \nAbstract: Large language models (LLMs) such as GPT-5 and Gemini 3 have pushed the frontier of automated reasoning and code generation. Yet current benchmarks emphasize accuracy and output quality, neglecting a critical dimension: efficiency of token usage. The token efficiency is highly variable in practical. Models solving the same problem with similar accuracy can exhibit up to a \\textbf{5.0$\\times$} difference in token length, leading to massive gap of model reasoning ability. Such variance exposes significant redundancy, highlighting the critical need for a standardized benchmark to quantify the gap of token efficiency. Thus, we introduce OckBench, the first benchmark that jointly measures accuracy and token efficiency across reasoning and coding tasks. Our evaluation reveals that token efficiency remains largely unoptimized across current models, significantly inflating serving costs and latency. These findings provide a concrete ro","title":"OckBench: Measuring the Efficiency of LLM Reasoning","url":"https://arxiv.org/abs/2511.05722","vendor":"arxiv_cs_ai"},"summary":"arXiv:2511.05722v3 Announce Type: replace-cross \nAbstract: Large language models (LLMs) such as GPT-5 and Gemini 3 have pushed the frontier of automated reasoning and code generation. Yet current benchmarks emphasize accuracy and output quality, neglecting a critical dimension: efficiency of token usage. The token efficiency is highly variable in practical. Models solving the same problem with similar accuracy can exhibit up to a \\textbf{5.0$\\times$} difference in token length, leading to massive gap of model reasoning ability. Such variance exposes significant redundancy, highlighting the critical need for a standardized benchmark to quantify the gap of token efficiency. Thus, we introduce OckBench, the first benchmark that jointly measures accuracy and token efficiency across reasoning and coding tasks. Our evaluation reveals that token efficiency remains largely unoptimized across current models, significantly inflating serving costs and latency. These findings provide a concrete ro","title":"OckBench: Measuring the Efficiency of LLM Reasoning","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-04T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2511.05722"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:0a64826d5fe72dbc2760240feb4ecd7892e480f5ad16b1746229a74240aa41f64f827b53235b84409fdf726c8f534e9b121e99109705a406bf07db58710bdd09","signer":"crovia.substrate","subject":{"observed_at":"2026-06-04T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2511.05722"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"b78829dfac2f0fb0f15177038407de7189a75f2fbc533755dbab785689f93169","leaf_index":212818,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fea7f2ae8a5bd9692760798e935a0b5192a76eec11257014556640cb12e5fe8d","side":"right"},{"sibling":"70780f70bc45031d461a9d7ca81e0dab494912b2c4db8282c1cba070f8af7968","side":"left"},{"sibling":"d908096942092f2db4ab08d31faa5a7937739e1f724188a2cc6dd99d80bd6af4","side":"right"},{"sibling":"1820cf8b4a90b67cb9cdaff90d8a0f0d7a128b46c80de3a6d5917eb62bce021e","side":"right"},{"sibling":"45410d05b96813c142cc13502a7952136d885c32876ba0f81d1e394fe2d156dc","side":"left"},{"sibling":"b043ef4c24a378c518cd937d028d5345784f2a0844965c6fb86a6480a4862c3d","side":"right"},{"sibling":"a4d720a21c128acd66da32542f5a6516eadd1108045e73ba02163baa0cf221fc","side":"left"},{"sibling":"5d10c089cc50333cc6f5ec8f5c37b251f9cdd4252a9604bb46ec83ac720abdd9","side":"right"},{"sibling":"b4e8a5af3a9fdc4bb815270cd541ee4baf09fb1d9e2cccd75c7346558e1f1e4c","side":"left"},{"sibling":"cfb460164a914d1f96a36aa17124b45bb5417829d2adbe5ca48375dbd842ec44","side":"left"},{"sibling":"a84ebc8e894a9893ee34d1afc942d15f28243b926f1e5f9d7f10a3de1e795262","side":"left"},{"sibling":"f8f6bd9da448fa097e2115b71146f61691d9f8807aca291c87c633192a7224e9","side":"left"},{"sibling":"2dca509b3eb767a47cf215d4315f230ce9103a76264412008ae23a349b519ef1","side":"left"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"422bcf7e281ca3a607f3726a5e6b8fabdb85e8b32199b4a86356998e260a0b34","side":"right"},{"sibling":"54a99163a4a62374c3ca6fb46294222f1d4b1a9d0b636e27256b0e093e98239a","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":213053,"merkle_root":"19d104b92c4d7299881c447fb8422611fc9cb8615d37a538959e34a2da7ef55f","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260604T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-04T05:37:49Z","sig_algorithm":"ed25519","signature":"630748e88645187aa3b4d4cb8c872cc146180d0301e4c6656f15f99081d7776432715cea2a997b32b7b3a5216f215d19c1b536c0b093100ec843fa0bd65f2101","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1a4649430b887c1bbafde15f4738a0344e04b5c77445297ea4cf2dea8e8cfa60"}}