{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_cadcc717503d038e9c90a3a0aa2971ef131011bbd554577504395a40d7e7087e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_cadcc717503d038e9c90a3a0aa2971ef131011bbd554577504395a40d7e7087e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"681f5cf08845d633277c6441d2cf92a51f7fe074634a6ee1dedf2a365d4205cf","published":"Fri, 26 Jun 2026 00:00:00 -0400","receipt_hash":"681f5cf08845d633277c6441d2cf92a51f7fe074634a6ee1dedf2a365d4205cf","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"681f5cf08845d633277c6441d2cf92a51f7fe074634a6ee1dedf2a365d4205cf","observed_at":"2026-06-26T04:43:58.168958Z","parent_run_hash":"9459505a803125e4b968df08c74ed0054a2aafd44e4e1a036e3b0709a8a65cb4","published":"Fri, 26 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.22437v2 Announce Type: replace-cross \nAbstract: We conduct a systematic study of 18 widely used vision-language benchmarks and identify three major issues: 1) many items do not rely on visual cues and therefore fail to effectively measure multimodal understanding; 2) many items are already close to performance saturation for current LVLMs, which limits their discriminative power; 3) a small number of anomalous items affect the reliability of evaluation results. To this end, we propose MMGist, a curated benchmark that covers seven capability dimensions and contains 7,262 items. MMGist is constructed through a three-stage pipeline, which sequentially combines text-ablation filtering, cross-model saturation filtering, and anomaly detection filtering. We conduct extensive experiments on 27 leading LVLMs and compare MMGist with the raw pool of 23,250 items. The results show that MMGist preserves model rankings with high fidelity, with Spearman $\\rho = 0.98$, while reducing evalua","title":"MMGist: A Comprehensive Multimodal Benchmark for 2027","url":"https://arxiv.org/abs/2606.22437","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.22437v2 Announce Type: replace-cross \nAbstract: We conduct a systematic study of 18 widely used vision-language benchmarks and identify three major issues: 1) many items do not rely on visual cues and therefore fail to effectively measure multimodal understanding; 2) many items are already close to performance saturation for current LVLMs, which limits their discriminative power; 3) a small number of anomalous items affect the reliability of evaluation results. To this end, we propose MMGist, a curated benchmark that covers seven capability dimensions and contains 7,262 items. MMGist is constructed through a three-stage pipeline, which sequentially combines text-ablation filtering, cross-model saturation filtering, and anomaly detection filtering. We conduct extensive experiments on 27 leading LVLMs and compare MMGist with the raw pool of 23,250 items. The results show that MMGist preserves model rankings with high fidelity, with Spearman $\\rho = 0.98$, while reducing evalua","title":"MMGist: A Comprehensive Multimodal Benchmark for 2027","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-26T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.22437"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:af5b019530776ec3d8efee3029a1094dbfa7122c8885db155b7665d407e1a9a847674a58698c6e1589fa014650935a9e6b1ccaa02a4596d9173a4e24532abc01","signer":"crovia.substrate","subject":{"observed_at":"2026-06-26T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.22437"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"35bd318852eff752c4bb94b7c30da1f11895baa900ac370065ae6fca5049f8c2","leaf_index":251259,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f35f22cf17325b0285ffdec439d262b22844575d1c1e90cf3e075f9eaeb9d6dd","side":"left"},{"sibling":"db05b0b5dc92cf20a1f651acb522f488924d6ba01ad43fcbf24f23c3d461bc6d","side":"left"},{"sibling":"e177aa714917616f5db0d55dec7d3ad238d8d05100f790e91cadc603e47d3e37","side":"right"},{"sibling":"6f8c7145f65330e6506f0ef3b11939c66209003b156bffd9829ba35b33f071b8","side":"left"},{"sibling":"973238270d3540b2e0c3af63d14225829dc18bfc2ceedb63cb72faeedcdabde7","side":"left"},{"sibling":"51263142806731dfcb141f012b12b625e80221cded7cd2fe31f3f94c663cb98d","side":"left"},{"sibling":"7de9e0cbfd7739efcc9d34b6ad0610c6607927b8ed4fcecf3bc616e114ce53c4","side":"left"},{"sibling":"846fe23f5401fbe8ae1dd175eade4cb955db0a1adde862418a9a781a48f6283c","side":"right"},{"sibling":"d097e1d1d25acb37e74792d46ad69ba853039e2f2b33a092c9ea32a837aeed84","side":"left"},{"sibling":"b6709caadc8510310ee2ec0b66d1058fcad31c91c65cc2bad6f46a693d553580","side":"right"},{"sibling":"a72c3b8804a37d1a9d18e02e6fdb048bc8ea6b0746909cd2de10bdbabc793737","side":"left"},{"sibling":"803703dc2c50a646fa77b57c0056e9a5126611ba4bcde0d6013ccd6b2d44bdbf","side":"right"},{"sibling":"e78f244b1b8df6d5e3fdc6dd76b5c27d4e6fe3b93b8cd61355497b63d7e4cfe8","side":"left"},{"sibling":"6167cb552ed6871fbf0afcf3db01d1017af7d472b136fcbe5404d1df09f41cc1","side":"right"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":251380,"merkle_root":"e042805d07cd8dc777d49695ad78b8d4ec9721df271ff0245d7706773c30b4a5","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260626T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-26T05:37:59Z","sig_algorithm":"ed25519","signature":"c68a6e827804771acd244208495c5e35c6f417307db3c76192f038dedaa5b5f86e019074a3bea353357ff7457924c4f7907832638bb9a4d3d18e7a7f50c6a10b","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_cadcc717503d038e9c90a3a0aa2971ef131011bbd554577504395a40d7e7087e"}}