{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_55ef39b29f3d9154ad7ef0e0aaf4cb6375d60092ec84279e5ef5fc6c97b8b572","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_55ef39b29f3d9154ad7ef0e0aaf4cb6375d60092ec84279e5ef5fc6c97b8b572","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"82905b8417144b5321c52fccbf20a2347802b456e7887e81a39593b3c14b5676","published":"Wed, 24 Jun 2026 00:00:00 -0400","receipt_hash":"82905b8417144b5321c52fccbf20a2347802b456e7887e81a39593b3c14b5676","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"82905b8417144b5321c52fccbf20a2347802b456e7887e81a39593b3c14b5676","observed_at":"2026-06-24T04:43:17.877668Z","parent_run_hash":"ca17d06d44ba7db934e6f913874699efc608b8f87453f1ac67f52060e620b57c","published":"Wed, 24 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.23712v1 Announce Type: cross \nAbstract: Audio-visual speech enhancement (AVSE) exploits visual cues such as lip movements to recover speech in noisy environments. Recent work introduced diffusion-based unsupervised AVSE, where a speech diffusion model conditioned on visual features via cross-attention is trained and used as a data-driven prior for posterior sampling-based speech enhancement. Despite promising performance over its audio-only counterpart, the impact of explicitly enforcing cross-modal alignment in the fusion remains unclear. In this work, we propose to augment the diffusion training objective with a contrastive audio-visual loss to encourage stronger use of visual information while keeping the posterior sampling framework unchanged. Experiments across matched and mismatched test data show consistent improvements in interference suppression, signal reconstruction, and perceptual quality, with the largest gains at low SNRs. Code is available at https://github.co","title":"Audio-visual Contrastive Alignment for Diffusion-based Visual-conditioned Speech Enhancement","url":"https://arxiv.org/abs/2606.23712","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.23712v1 Announce Type: cross \nAbstract: Audio-visual speech enhancement (AVSE) exploits visual cues such as lip movements to recover speech in noisy environments. Recent work introduced diffusion-based unsupervised AVSE, where a speech diffusion model conditioned on visual features via cross-attention is trained and used as a data-driven prior for posterior sampling-based speech enhancement. Despite promising performance over its audio-only counterpart, the impact of explicitly enforcing cross-modal alignment in the fusion remains unclear. In this work, we propose to augment the diffusion training objective with a contrastive audio-visual loss to encourage stronger use of visual information while keeping the posterior sampling framework unchanged. Experiments across matched and mismatched test data show consistent improvements in interference suppression, signal reconstruction, and perceptual quality, with the largest gains at low SNRs. Code is available at https://github.co","title":"Audio-visual Contrastive Alignment for Diffusion-based Visual-conditioned Speech Enhancement","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-24T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.23712"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:6b971c2e730e962db893fc091af32e1bbeb82c252a5865a88ceebfd92d1cd085182fce908764660d5fce8f2d55e374a6716276742c76dc8eb08397ddb3093a01","signer":"crovia.substrate","subject":{"observed_at":"2026-06-24T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.23712"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"be059fa4c8f69f3701a6010fab91d9c4605563aa577b0456c04f86efe88308ab","leaf_index":244493,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"393a056f6f727d37fb9d3c8b959d0838e949a4c433ec831c1408f66e28abcf6c","side":"left"},{"sibling":"b59a63dfb5863f771ce8026189972981db34553f9ca680add0c301082840e7f9","side":"right"},{"sibling":"fe46fc960baecdb7c014592af50c192cb7f10fb6d5fe2a52637e270817a08d4b","side":"left"},{"sibling":"f438b5132af2a6b8c58fb680a680ac8a2fd1f2019f3049a5c3b87a0bf4a9c293","side":"left"},{"sibling":"d0282a8cff839c0e4fea98e92d98d03673d14698043d8e787e6debc20f307c03","side":"right"},{"sibling":"d7a80dbc109a440cbbc5ee38ada545b76d1b74e87a0a9424159972f613d5d8b8","side":"right"},{"sibling":"f6c01d42388f96d80ab8ec2f12119f28971403b41d996362c6d78ca0e42293de","side":"right"},{"sibling":"d4fcf6bcc7f1fe78997b28df74b23bba21f5b2585d6fe8efb802f0d067291d5c","side":"right"},{"sibling":"0fa23771b702ff726ed1fc5a44f9b416b2a7861c2f957fcac2d95392276d4784","side":"left"},{"sibling":"bc74ebb08462da8a50fc65ea75f8a8a3418d10ebd471d830f1c67f33dd54dfd1","side":"left"},{"sibling":"6dafd355e5d54c60e61c6c02d3842984e234b1f5bca1623fcdd3def7b8931973","side":"right"},{"sibling":"3107b9d4dbf9456a39f99de694a4dd4da2c0600f9f8855f125161335fe8810af","side":"left"},{"sibling":"86118ab4500c3055a2af70062751a960423c464405b18ca1c37411bf0ce3f52e","side":"left"},{"sibling":"3a42039065acac6d3e4088ec61d9c116ecf7a26c7b7163d23da8fd0b3362e038","side":"left"},{"sibling":"c044f2bd864a0e8e8af5a7f6e3124def7fc4b4511b2b166ea8f9de321e8d385e","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":244827,"merkle_root":"274e133c6dfa2781a9cfb85337d01cc6b72688ce5e810149f3183e400ffab136","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260624T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-24T05:37:55Z","sig_algorithm":"ed25519","signature":"22ca3cee4de2447b3d281e30e09fe566461996bb7be4d4465f08a3f4cf59cea58f22683a4aa10ef4d5d17a19b03f4212392bfd26f2b51f289f0cdf1042a03800","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_55ef39b29f3d9154ad7ef0e0aaf4cb6375d60092ec84279e5ef5fc6c97b8b572"}}