{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_797368aa172f8e80c0576ff8aecb331c091bbe391c8334cc2b1160f9ac0a6bc7","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_797368aa172f8e80c0576ff8aecb331c091bbe391c8334cc2b1160f9ac0a6bc7","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"e0e10198d17ef26f4b346d55f2e8ce0c983eb2ddf76c1e2dcdb263e1c16d082c","published":"Fri, 10 Jul 2026 00:00:00 -0400","receipt_hash":"e0e10198d17ef26f4b346d55f2e8ce0c983eb2ddf76c1e2dcdb263e1c16d082c","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"e0e10198d17ef26f4b346d55f2e8ce0c983eb2ddf76c1e2dcdb263e1c16d082c","observed_at":"2026-07-10T04:43:53.465232Z","parent_run_hash":"06997be187ba20932a2030c56de194579eacf484085252a25bee544eab183e91","published":"Fri, 10 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.08317v1 Announce Type: new \nAbstract: Modern AI models achieve strong performance on many established benchmarks, yet they still fail on tasks that humans find almost trivial, such as manipulating a string or drawing a dog with five legs. These examples suggest that existing benchmarks may under-measure persistent blind spots in current systems. We introduce $\\texttt{blind-spots-bench}$, a benchmark designed to expose such blind spots through tasks that appear simple for humans but remain challenging for modern AI. We collect raw questions from students in an AI course, clean and annotate them with structured reference solutions, and propose a task taxonomy tailored to the resulting dataset of 235 samples. We further develop an automated grading pipeline to evaluate a wide range of models, including open-weight and closed-source language, vision-language, and image-generation models. Our analysis on $\\texttt{blind-spots-bench}$ reveals that closed-source frontier models can ","title":"Blind-Spots-Bench: Evaluating Blind Spots in Multimodal Models","url":"https://arxiv.org/abs/2607.08317","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.08317v1 Announce Type: new \nAbstract: Modern AI models achieve strong performance on many established benchmarks, yet they still fail on tasks that humans find almost trivial, such as manipulating a string or drawing a dog with five legs. These examples suggest that existing benchmarks may under-measure persistent blind spots in current systems. We introduce $\\texttt{blind-spots-bench}$, a benchmark designed to expose such blind spots through tasks that appear simple for humans but remain challenging for modern AI. We collect raw questions from students in an AI course, clean and annotate them with structured reference solutions, and propose a task taxonomy tailored to the resulting dataset of 235 samples. We further develop an automated grading pipeline to evaluate a wide range of models, including open-weight and closed-source language, vision-language, and image-generation models. Our analysis on $\\texttt{blind-spots-bench}$ reveals that closed-source frontier models can ","title":"Blind-Spots-Bench: Evaluating Blind Spots in Multimodal Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-10T04:43:53Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.08317"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:f2a80501496f97e93053e79dbf74c87bfa24e13aaf12d8932ee4c03d348b3c311aed4c0bd85d30773fa7e1d0d3896748fd5662d8247077dae9e141a6f7911b03","signer":"crovia.substrate","subject":{"observed_at":"2026-07-10T04:43:53Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.08317"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"dd9217b8a517dd817328582f2b57b798a2e5489d2c953652b8bcf928d9289b84","leaf_index":299387,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"1bab230ebd17225d7a75a2866c1e42ff338ee9cd3be21e32f84c019367a4f6f5","side":"left"},{"sibling":"505e44cd3878d9727cebe1c956cf7de57e9d553692ee18edf64923dcf1cb480c","side":"left"},{"sibling":"6af23dfd7c6d7302aa2f5f4d3ec70b78a06a48714fe8709c710468fe1b43e584","side":"right"},{"sibling":"9f3c44c51e54baddd0b1ef4ca78a62404f74e18dffae23d7b856ba3901127698","side":"left"},{"sibling":"b97a180fea776ff338aaf504831d817010d3ee08552cc6583c2b041679dd8b38","side":"left"},{"sibling":"f8d66ecc290a865e1dc8caf3d9a9176b9cc0743edec77de9e8ca456b72551a4c","side":"left"},{"sibling":"d7d1a57010639fd7da98ae2622b06b3bf73287b1959dbdbc7fa2102045b05bc1","side":"left"},{"sibling":"442ba9ebe9175420bc5e043f85ccde9011456ea96aa073c7096ba1d5b28a3706","side":"right"},{"sibling":"2c5ce75ad134aa442d7fbb5dc9968b0f2e48097c2ffaa2111d405de1c4a9c48b","side":"left"},{"sibling":"36a2c507282befad57202dd10278a66d37b402692f195a22d2275b3d1b2488d8","side":"right"},{"sibling":"f80e8d47e0860527b906fc2dba9a52609f7f623f2772479ae922cc019bab36d9","side":"right"},{"sibling":"cae83500ab2c25555aa6b5eaf9232696d15a868d91b34f7531dd955daadf70f7","side":"right"},{"sibling":"64dab64d51bdcb909e2a5e37efb8909d6704ecf824be484b5d2b60ee6e518890","side":"left"},{"sibling":"576f134a23c19a758ae5efd53016092a74b9900e878cf6eb4f3dab6be682b395","side":"right"},{"sibling":"3c65f53d7c3e4feba7c745e8df1327760ffa768eec84336db14d515a31731532","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"8728642cdb98496d916cc653f0d919d7eb2e89d0c529927c9e89091074ad584c","side":"right"},{"sibling":"8025674cb002a22ae243ca0c295c18c1d0ee119189ea88e08ac14a3a1468b8e3","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":299721,"merkle_root":"f7115d63193d3285ca28cb9f741ecea2513f9b3e492e785f97076f3cf8f9bb98","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260710T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-10T05:38:19Z","sig_algorithm":"ed25519","signature":"4eeedeb744885bff6523d66b1cb86bde62b36917696367addfd980d90b31011daece2e01b20608e4c0870ec31dd0bcb57d297447f01f153bf0716ad527142b00","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_797368aa172f8e80c0576ff8aecb331c091bbe391c8334cc2b1160f9ac0a6bc7"}}