{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_c700adad0fcd74a804d00d1b59beb18eccb6a154bbbeffd65d0438935a8d55c1","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_c700adad0fcd74a804d00d1b59beb18eccb6a154bbbeffd65d0438935a8d55c1","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6715cfe6f8c34725c1a51c4608806542b3868a0ee8e3eca2cb8d2cc44eddc3c6","published":"Fri, 22 May 2026 00:00:00 -0400","receipt_hash":"6715cfe6f8c34725c1a51c4608806542b3868a0ee8e3eca2cb8d2cc44eddc3c6","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6715cfe6f8c34725c1a51c4608806542b3868a0ee8e3eca2cb8d2cc44eddc3c6","observed_at":"2026-05-22T04:43:12.593252Z","parent_run_hash":"dd4d56660d55b2d65dc84dc5e7c8f83487d90da2dd343b5707d6948a3bb0d917","published":"Fri, 22 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2603.28103v2 Announce Type: replace-cross \nAbstract: Parliamentary proceedings represent a rich yet challenging resource for computational analysis, particularly when preserved only as scanned historical documents. Existing efforts to transcribe Italian parliamentary speeches have relied on traditional Optical Character Recognition pipelines, resulting in transcription errors and limited semantic annotation. In this paper, we propose a pipeline based on Vision-Language Models for the automatic transcription, semantic segmentation, and entity linking of Italian parliamentary speeches. The pipeline employs a specialised OCR model to extract text while preserving reading order, followed by a large-scale Vision-Language Model that performs transcription refinement, element classification, and speaker identification by jointly reasoning over visual layout and textual content. Extracted speakers are then linked to the Chamber of Deputies knowledge base through SPARQL queries and a mult","title":"Transcription and Recognition of Italian Parliamentary Speeches Using Vision-Language Models","url":"https://arxiv.org/abs/2603.28103","vendor":"arxiv_cs_ai"},"summary":"arXiv:2603.28103v2 Announce Type: replace-cross \nAbstract: Parliamentary proceedings represent a rich yet challenging resource for computational analysis, particularly when preserved only as scanned historical documents. Existing efforts to transcribe Italian parliamentary speeches have relied on traditional Optical Character Recognition pipelines, resulting in transcription errors and limited semantic annotation. In this paper, we propose a pipeline based on Vision-Language Models for the automatic transcription, semantic segmentation, and entity linking of Italian parliamentary speeches. The pipeline employs a specialised OCR model to extract text while preserving reading order, followed by a large-scale Vision-Language Model that performs transcription refinement, element classification, and speaker identification by jointly reasoning over visual layout and textual content. Extracted speakers are then linked to the Chamber of Deputies knowledge base through SPARQL queries and a mult","title":"Transcription and Recognition of Italian Parliamentary Speeches Using Vision-Language Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-22T04:43:12Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2603.28103"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:460417b15fb62e27ae438fd9f586cc00a48b840a542ce01d2de477e4797beb16066dc8d9051fa7e31186e95b8c2794741b1d97d198e34792f5a16ec0d2524202","signer":"crovia.substrate","subject":{"observed_at":"2026-05-22T04:43:12Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2603.28103"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"57d5f300b405342e7fad86cde18c0c8917c54d2180ae8c773805e6766ccec5c1","leaf_index":148458,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"e894ea44eb069eb9b4251da7c54fa24327cebdfa88d56785882de57282c5eb5b","side":"right"},{"sibling":"8e75d9b20077a4e92488b65899bde948d666ef2151f74d9de180666b397110a5","side":"left"},{"sibling":"ba3142a626a718744700d2c60b97e39eb465e6e19cd6f1ae1d5bf46991fa49b1","side":"right"},{"sibling":"86275baf394b75e531ebfe1de3b0964c4348a2b63023c54c3333a3e78dc4d8b9","side":"left"},{"sibling":"3a9366dc48648639578cc1b8c2bf3601a7b1146aceab5b0cc886155d77874bec","side":"right"},{"sibling":"aac0a67d982a8a56aa1e6c1a7e85d296a7c628706001055d4cd4a1198acfde64","side":"left"},{"sibling":"aceda386eeb461be8d916aae46f429d11515d6911251d4a2391679ac476d82b5","side":"left"},{"sibling":"753da9fa5ffb91dd383318229c9999d218b1611b357943fc0739373f06f55d17","side":"left"},{"sibling":"e6aaada16cbe58b790855c6723700fd07821cc6df59048dc470d404b1ba60842","side":"left"},{"sibling":"d337a9fdcfc121e9691d8db9173af6a3fe0c33d4a6d5f0c8a7a01af04f9fb856","side":"left"},{"sibling":"8b39e07457f5cc5d687d2ae42284dbe705bb87084e7db4b626aff81e51dacd19","side":"right"},{"sibling":"79a713e1e345ccb99c5fe994a11708c8e9bcfa2e91f940d70421cb7d8d77ecc6","side":"right"},{"sibling":"249870fb494bef050c409081e5de45f9042938d7b2823524ee496f296aa63667","side":"right"},{"sibling":"96c48ee8328f1b7925a4cc4421df5cb0bd81c92a5d8354c93126fa5f0166d225","side":"right"},{"sibling":"35ca36cee447f0ef7064a25d55f59357c66901e31427729c4c1d8b14aa8adb6c","side":"left"},{"sibling":"4e13b4a3e69bb83d13913477a782913ce03937edd65046853c1964d3cbb6564b","side":"right"},{"sibling":"0f7b2df1c4580bf7bb7c24b9158ba20593a06af18a5f18d0973e5eff20c35cd8","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":148601,"merkle_root":"44900cffd986f40535c83f46a250e86fbf1019d41f27080a00fbf9b8d77ec33a","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260524T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-24T13:37:32Z","sig_algorithm":"ed25519","signature":"059e428c3c5241de303721ad6ac7b748758372180f3f0a717810312aecd6fab073abb3ea264157666de581a2b361c4c6e2aa8ca081b08ce9be090721f0e3400e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_c700adad0fcd74a804d00d1b59beb18eccb6a154bbbeffd65d0438935a8d55c1"}}