{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_5133cecec2c7580ac96bfd70c93eb2bd858c0f501ec17f4ea4e8d11397a73664","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_5133cecec2c7580ac96bfd70c93eb2bd858c0f501ec17f4ea4e8d11397a73664","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"bff0f3d9c6911bf5f1f4d12533c664f92084d58951d8a3ef134735896bf58675","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"bff0f3d9c6911bf5f1f4d12533c664f92084d58951d8a3ef134735896bf58675","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"bff0f3d9c6911bf5f1f4d12533c664f92084d58951d8a3ef134735896bf58675","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2604.07593v2 Announce Type: replace \nAbstract: Mathematical benchmarks consisting of a range of mathematics problems are widely used to evaluate the reasoning abilities of large language models, yet little is known about how their structural properties influence model behaviour. In this work, we investigate two structural length variables, prompt length and solution length, and analyse how they relate to model performance on a newly constructed adversarial dataset of expert-authored mathematics problems. We find that both prompt and solution lengths correlate positively with increased model failure across models. We also include a secondary, exploratory analysis of cross-model disagreement. Under a difficulty-adjusted normalised analysis, both variables retain weak negative associations with realised model separation, slightly stronger for prompt length. Overall, our main robust finding is that structural length is linked to empirical difficulty in this dataset.","title":"Too long; didn't solve","url":"https://arxiv.org/abs/2604.07593","vendor":"arxiv_cs_ai"},"summary":"arXiv:2604.07593v2 Announce Type: replace \nAbstract: Mathematical benchmarks consisting of a range of mathematics problems are widely used to evaluate the reasoning abilities of large language models, yet little is known about how their structural properties influence model behaviour. In this work, we investigate two structural length variables, prompt length and solution length, and analyse how they relate to model performance on a newly constructed adversarial dataset of expert-authored mathematics problems. We find that both prompt and solution lengths correlate positively with increased model failure across models. We also include a secondary, exploratory analysis of cross-model disagreement. Under a difficulty-adjusted normalised analysis, both variables retain weak negative associations with realised model separation, slightly stronger for prompt length. Overall, our main robust finding is that structural length is linked to empirical difficulty in this dataset.","title":"Too long; didn't solve","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2604.07593"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ebac6148327596f3754fba79803e322d583e28a9378d322d041458d2da56eda7f56397b50703bc902d39f88c5e0492ade699b6e6baa6463abb6595e1ac831009","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2604.07593"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"ef5406673725ca5288a86f6fd9713001eaadcaada7418129ce0b7f62a8fd17ed","leaf_index":235690,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"9bc0b2fda34e001240e2d1361a8679bd342675aadfbdc7fafb668b6a6ed59ac9","side":"right"},{"sibling":"86fb9da880dbdc17f85d56ff6511537165927d7184f030fb552979384fee00d7","side":"left"},{"sibling":"19a49a4506a513a7a0f58408df0c296fa1035b23d7cdcf28c665657b78865131","side":"right"},{"sibling":"bd20ec4ba0914185eaba3355163c18323fa6ad9750efe71c06cd58ce3d260f20","side":"left"},{"sibling":"13b456264d993510a1862fbe736fe444cc9ce0e25d3a03c05ab459c12ec4e13e","side":"right"},{"sibling":"7dc6760aab106e939aa5db80f98209b84ff8e42be64c8d3a64855d28698ac2fc","side":"left"},{"sibling":"d4e4da8292a59c15443bbd4cd8c53ca2bf82cd90b34e48fb7704c5e2e86fe09f","side":"right"},{"sibling":"9af7ea2b04101a414ff44ef903d5d381777f0286b489347e7e959b64158ff697","side":"left"},{"sibling":"aa68ebe8f5e8e96388fc8d1af3aa08be7ccd27ab4cebcf5913560c48d877bc27","side":"right"},{"sibling":"8253d44cf1ed30d3ab19c2b339fb4000a1fa173182c65390e9e8dabf8173b9e9","side":"right"},{"sibling":"e2bf9b60400244c698c0196109f54323457abc5b64dee08ec33ab14cc4faaef7","side":"right"},{"sibling":"86664e7f68ba08b8dfcf77dda51a4dfa7fcfc986d4ad7c704ffb71b669202da7","side":"left"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_5133cecec2c7580ac96bfd70c93eb2bd858c0f501ec17f4ea4e8d11397a73664"}}