{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_66ce49110058df7a3b04ff02cdfdee796bbde0edca278a5d0d6f17f4db2c43ca","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_66ce49110058df7a3b04ff02cdfdee796bbde0edca278a5d0d6f17f4db2c43ca","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6d076cfcc1b9a59a04a3511c54d473b7afb77e44f4a481e6fbdc680031220271","published":"Fri, 19 Jun 2026 00:00:00 -0400","receipt_hash":"6d076cfcc1b9a59a04a3511c54d473b7afb77e44f4a481e6fbdc680031220271","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6d076cfcc1b9a59a04a3511c54d473b7afb77e44f4a481e6fbdc680031220271","observed_at":"2026-06-19T04:43:39.497162Z","parent_run_hash":"942f204649bd8fb7e5f3ac68f64dc64a5a02624b49ac200c0f629f6ff3a211f3","published":"Fri, 19 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.19534v1 Announce Type: cross \nAbstract: Multimodal large language models (MLLMs) have achieved remarkable progress in visual understanding tasks. However, most existing MLLMs rely on autoregressive generation, which limits their efficiency for perception tasks that require captioning multiple regions. In this work, we propose PerceptionDLM, a multimodal diffusion language model optimized for efficient parallel region perception. Built upon PerceptionDLM-Base, a strong foundational baseline that achieves state-of-the-art performance among open-source diffusion MLLMs, our architecture fully leverages the parallel decoding nature of DLMs. Specifically, we introduce efficient prompting and structured attention masking to enable simultaneous perception of multiple masked regions, allowing the model to generate region descriptions in parallel at both the sequence and token levels. This design significantly improves inference efficiency compared with existing approaches that proces","title":"PerceptionDLM: Parallel Region Perception with Multimodal Diffusion Language Models","url":"https://arxiv.org/abs/2606.19534","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.19534v1 Announce Type: cross \nAbstract: Multimodal large language models (MLLMs) have achieved remarkable progress in visual understanding tasks. However, most existing MLLMs rely on autoregressive generation, which limits their efficiency for perception tasks that require captioning multiple regions. In this work, we propose PerceptionDLM, a multimodal diffusion language model optimized for efficient parallel region perception. Built upon PerceptionDLM-Base, a strong foundational baseline that achieves state-of-the-art performance among open-source diffusion MLLMs, our architecture fully leverages the parallel decoding nature of DLMs. Specifically, we introduce efficient prompting and structured attention masking to enable simultaneous perception of multiple masked regions, allowing the model to generate region descriptions in parallel at both the sequence and token levels. This design significantly improves inference efficiency compared with existing approaches that proces","title":"PerceptionDLM: Parallel Region Perception with Multimodal Diffusion Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-19T04:43:39Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.19534"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:643320087ef79f382bd9e169c771ea96ccd1846d4c8099df99c504608d0e6e031078e5272293be8fe7254bef86e8f7361e81380de8fdef2f56cf5ae02bd1a70c","signer":"crovia.substrate","subject":{"observed_at":"2026-06-19T04:43:39Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.19534"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7e7bebe15bd89b60b296d977cdba74752be87a8d30c72a3ae3f70f1844c14113","leaf_index":235565,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"a29deffeaf51fd60149215681bab186439991b3874bd08bca519bca9b02a60aa","side":"left"},{"sibling":"7a72ec270b93d1143e39bd9d50c2dc60018ebeedbfb2c5db6dfc26d8afbcb5d2","side":"right"},{"sibling":"aa8e8d7d10843f0587f9fb25fcb4beedf39a4e9a3d36f916e4673df3f9fb9e8c","side":"left"},{"sibling":"2bb110b8a28db8076a9784f64f5c237890961db92b15f1db026410f149c9468f","side":"left"},{"sibling":"916963b4e1c2a7a0d208da26eef17d0573fb38fcb0f1b14ca61daf47d61582a0","side":"right"},{"sibling":"c51c58a50c9136814d4c77707235d4cb07ccd1952c8d9530f44c0a4e8d56c7b8","side":"left"},{"sibling":"6cc7ca0021a188fb7ef3325d99f198a5bb7782a033e787d0adde01b21a90e6f5","side":"right"},{"sibling":"c0ab0c7dd98a2c17ee965cb19831d49d3a82a2ba1812d8a677f78274ab6f7c98","side":"right"},{"sibling":"aa68ebe8f5e8e96388fc8d1af3aa08be7ccd27ab4cebcf5913560c48d877bc27","side":"right"},{"sibling":"8253d44cf1ed30d3ab19c2b339fb4000a1fa173182c65390e9e8dabf8173b9e9","side":"right"},{"sibling":"e2bf9b60400244c698c0196109f54323457abc5b64dee08ec33ab14cc4faaef7","side":"right"},{"sibling":"86664e7f68ba08b8dfcf77dda51a4dfa7fcfc986d4ad7c704ffb71b669202da7","side":"left"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"b52a771530dd1686bca49e42088898b86da94879579cd6a995c6ab0598a665fe","side":"right"},{"sibling":"a116bb92f9b0350491155b470acc86d006c33ec558759e49e56614a54c39f242","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":241122,"merkle_root":"7a906c6a26ff6c6feabc2feaba6a1a70c515e6fd72a38c779293b0f78ff291c4","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260622T183701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-23T06:25:25Z","sig_algorithm":"ed25519","signature":"5576b1d56d5dbb0d96c780fa3ca0940d805c8de95c6251bc87297f0be058aa5e37eb53a6aa1b601381f489f093842cf674b28737ed8e46ce3a49814b5e57290c","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_66ce49110058df7a3b04ff02cdfdee796bbde0edca278a5d0d6f17f4db2c43ca"}}