{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_890ae225ac9f408fe281c76477c832719ca7e5573966a797c173953a7dc37a7e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_890ae225ac9f408fe281c76477c832719ca7e5573966a797c173953a7dc37a7e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"1e8e847e447ada00cf40d70ce34dcc9412900f135f2402f2feb4fb2244c36321","published":"Wed, 20 May 2026 00:00:00 -0400","receipt_hash":"1e8e847e447ada00cf40d70ce34dcc9412900f135f2402f2feb4fb2244c36321","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"1e8e847e447ada00cf40d70ce34dcc9412900f135f2402f2feb4fb2244c36321","observed_at":"2026-05-20T04:43:44.562035Z","parent_run_hash":"5f904c2c2fecb6b44f2adce8bdc9de914b9a39f7c4fdd7bf086a6c2361a30f8c","published":"Wed, 20 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2505.17726v3 Announce Type: replace-cross \nAbstract: Recently, multimodal large language models (MLLMs) have emerged as a key approach in achieving artificial general intelligence. In particular, vision-language MLLMs have been developed to generate not only text but also visual outputs from multimodal inputs. This advancement requires efficient image tokens that LLMs can process effectively both in input and output. However, existing image tokenization methods for MLLMs typically capture only global abstract concepts or uniformly segmented image patches, restricting MLLMs' capability to effectively understand or generate detailed visual content, particularly at the object level. To address this limitation, we propose an object-centric visual tokenizer based on Slot Attention specifically for MLLMs. In particular, based on the Q-Former encoder, diffusion decoder, and residual vector quantization, our proposed discretized slot tokens can encode local visual details while maintaini","title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","url":"https://arxiv.org/abs/2505.17726","vendor":"arxiv_cs_ai"},"summary":"arXiv:2505.17726v3 Announce Type: replace-cross \nAbstract: Recently, multimodal large language models (MLLMs) have emerged as a key approach in achieving artificial general intelligence. In particular, vision-language MLLMs have been developed to generate not only text but also visual outputs from multimodal inputs. This advancement requires efficient image tokens that LLMs can process effectively both in input and output. However, existing image tokenization methods for MLLMs typically capture only global abstract concepts or uniformly segmented image patches, restricting MLLMs' capability to effectively understand or generate detailed visual content, particularly at the object level. To address this limitation, we propose an object-centric visual tokenizer based on Slot Attention specifically for MLLMs. In particular, based on the Q-Former encoder, diffusion decoder, and residual vector quantization, our proposed discretized slot tokens can encode local visual details while maintaini","title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-20T04:43:44Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2505.17726"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:5b004cb61c7c9002853425d47cd15dffec5a3d9478aa556884c549f39201c8010d5483a29ad1a74c175ce85c8d27a2fa414a032126691026ccc3f4f074ce1c0f","signer":"crovia.substrate","subject":{"observed_at":"2026-05-20T04:43:44Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2505.17726"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f8c32077fafbd784ba81191ab8a7850d61a6154a209f5b5da194661b8180cc08","leaf_index":145249,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"2f8bd2eac46ba4338be9ab4fca04585c12ebe055833ded7f294a3b3af24e80c5","side":"left"},{"sibling":"83c23071c199a0d3f8953825fd119ddc70cd3772f4f87300c74f0e49134874d6","side":"right"},{"sibling":"123d2e77b443106b313e133afe14b86a3e679d210bb5a6d501767b66886fb854","side":"right"},{"sibling":"a4d64dbbe9cef549eafd957b9aa6d67b616b52029a4bb438b2731d7e0aa79c7e","side":"right"},{"sibling":"081e6c3b74d7fc7fae6ba1fba27a92ee6d227f65bebd759e46e1c754e1634b78","side":"right"},{"sibling":"b891f8102dbec11b5872d47b2e640beca78f5f15b3e6d01c8d7bbebe1e3b209f","side":"left"},{"sibling":"8abfb0c0cfeb30fba16eb01548a8b7ce35f077e44d727cc0fe4a62d471800c6e","side":"left"},{"sibling":"79fb6d27e8a49dd15a97fca96f52d62ff5ee7213d44a5526459d33f42a050928","side":"right"},{"sibling":"f073aa7be27ee7e1d9eb0de5f129f7e9bae0c584fb959378c395b65831dc1d1d","side":"left"},{"sibling":"1526885f19d1fadf6955cf519dbc4e62d593a4bba99d741e8c301740a7068233","side":"left"},{"sibling":"e3a7d5c07f161682d61bd453ffc02ecdf87cfeda70f986d6650017f9d2d6b265","side":"left"},{"sibling":"edbc49f08e5b92291934c05c9e6efd270a6b0698d8d2fa474006366027dfe098","side":"right"},{"sibling":"3e4df6e7457cecbf36f350375e72dcab336a3984422e4c406ef809e4e2944e96","side":"left"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"be08fedc4e72a6fb56606f66812fae7317e09690b9acb18385f4ab117a981238","side":"right"},{"sibling":"0534329a7475dc9df51998c83c16892126679dade0fa34182f21e869599386c7","side":"right"},{"sibling":"2d24720928ead0e7670650eb55f558c4f20e4c18df376f47ba72cfa8cf0ed344","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":147301,"merkle_root":"08903d7159c3b38eeeeafc09eab15139ea417f1d94f02f1fbc87296b37db840a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260521T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-21T18:37:33Z","sig_algorithm":"ed25519","signature":"905f2924632dfa2970c8690285f5b5d4a1d891d0e0ef1cbc404ebec2fd937215ac768e16f0a9f28b18977a55ae0bfd226db6833ae7ef588729054117d2da7303","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_890ae225ac9f408fe281c76477c832719ca7e5573966a797c173953a7dc37a7e"}}