{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d801c0bf7c5bf9ff9b9052ef1fa5e3efd2bbacd8e0999dfb6df3b2290cc1d95d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d801c0bf7c5bf9ff9b9052ef1fa5e3efd2bbacd8e0999dfb6df3b2290cc1d95d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"b2b27338843a5630e96b08701bf370b6bd340db582d11f94e1a89495487a06a7","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"b2b27338843a5630e96b08701bf370b6bd340db582d11f94e1a89495487a06a7","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"b2b27338843a5630e96b08701bf370b6bd340db582d11f94e1a89495487a06a7","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.13178v2 Announce Type: replace-cross \nAbstract: In large vision-language models, visual tokens typically constitute the majority of input tokens, leading to substantial computational overhead. To address this, recent studies have explored pruning redundant or less informative visual tokens for image understanding tasks. However, these methods struggle with pixel grounding tasks, where token importance is highly contingent on the input text. Through an in-depth analysis of CLIP, we observe that visual tokens within referent regions often exhibit low similarity to their textual representation. Motivated by this insight, we introduce LiteLVLM, a training-free, text-guided token pruning strategy for efficient pixel grounding inference. By reversing the ranking of CLIP's visual-text similarity, LiteLVLM effectively retains visual tokens covering the referent regions, while recovering context tokens to enable clear foreground-background separation. Extensive experiments demonstrat","title":"CLIP Tricks You: Training-free Token Pruning for Efficient Pixel Grounding in Large VIsion-Language Models","url":"https://arxiv.org/abs/2605.13178","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.13178v2 Announce Type: replace-cross \nAbstract: In large vision-language models, visual tokens typically constitute the majority of input tokens, leading to substantial computational overhead. To address this, recent studies have explored pruning redundant or less informative visual tokens for image understanding tasks. However, these methods struggle with pixel grounding tasks, where token importance is highly contingent on the input text. Through an in-depth analysis of CLIP, we observe that visual tokens within referent regions often exhibit low similarity to their textual representation. Motivated by this insight, we introduce LiteLVLM, a training-free, text-guided token pruning strategy for efficient pixel grounding inference. By reversing the ranking of CLIP's visual-text similarity, LiteLVLM effectively retains visual tokens covering the referent regions, while recovering context tokens to enable clear foreground-background separation. Extensive experiments demonstrat","title":"CLIP Tricks You: Training-free Token Pruning for Efficient Pixel Grounding in Large VIsion-Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.13178"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:1f6adde9f96fe73be949b831d326bcdcb99eb2c7d55f628db1d745b178695bc109195cc8ae69fae05e5591adda54044aaa8cc1622f3ee0c342fef356e45ee605","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.13178"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7c1e835db1fac43d44ea07fa6c88ff72066763474d14398cfcfe244fcc67f8ef","leaf_index":206062,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"e19cc8c3c4df82a47dbbbc21620c0300db60944d4f332a0d938781accb35081c","side":"right"},{"sibling":"b6b1e6ba39e2d519e93ad3358475907cd21302a06e85d66385fa66b3f826ec55","side":"left"},{"sibling":"0c736c344a6154c445873b668f0b4333246d3c9de346f979db84b88e00f3f95e","side":"left"},{"sibling":"b4c82c176a1911a321ac7446504cd9071b71c6789d320c0197d35caff179a18e","side":"left"},{"sibling":"e1f61cf71ce80f3e38166c1dabc0884d9ac946087d78c2ec4b1c56e2bad01219","side":"right"},{"sibling":"9ac8388fe62227fcc09b0a753dd995a168cd39a8b8979b0166f1ec6a499ac280","side":"left"},{"sibling":"10ada48f52a601d26daa96f2bfeb4f348d3b5a39063f8e6ae49d78113ad822bd","side":"left"},{"sibling":"a87881f2c8c65f06a7d79e20af3b2022c798d3fe9dfa79b997557c49e1803708","side":"left"},{"sibling":"6ac554ede1f78dc3a28ebe5e1155c2b9c258b71805fdd7ad2ec992697f3ccd0c","side":"right"},{"sibling":"5e1949edfad76bc6a73d008dc8be8a0c6fe4a7fb64c8beaf70e5023ca11d35e0","side":"right"},{"sibling":"e4bd1aaaf3f336d9b072b4d1fc234246fc3cf2874ca873b7741c03eaf97911f2","side":"left"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d801c0bf7c5bf9ff9b9052ef1fa5e3efd2bbacd8e0999dfb6df3b2290cc1d95d"}}