{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_d714e5c7c87c9757e67b43dedd7a5048811e9c0c6d92e25d1b9569cdd3bd0d82","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_d714e5c7c87c9757e67b43dedd7a5048811e9c0c6d92e25d1b9569cdd3bd0d82","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a086c7cc7c7f58d3d903158a4b0f5e9ccb4900c6b413c11760a81fb7df83844b","published":"Fri, 26 Jun 2026 00:00:00 -0400","receipt_hash":"a086c7cc7c7f58d3d903158a4b0f5e9ccb4900c6b413c11760a81fb7df83844b","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a086c7cc7c7f58d3d903158a4b0f5e9ccb4900c6b413c11760a81fb7df83844b","observed_at":"2026-06-26T04:43:58.168958Z","parent_run_hash":"9459505a803125e4b968df08c74ed0054a2aafd44e4e1a036e3b0709a8a65cb4","published":"Fri, 26 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.26794v1 Announce Type: cross \nAbstract: CLIP and its variants are widely adopted visual backbones in multimodal systems, but their pretraining remains dominated by descriptive image-text alignment. As downstream applications increasingly demand visually grounded commonsense inference and compositional reasoning, it remains unclear whether CLIP-style encoders can support such reasoning without architectural changes. To address this, we present ReasonCLIP-58M, a continual pretraining framework that integrates large-scale reasoning supervision into CLIP-style models through our two-stage strategy, which progressively integrates reasoning signals while preserving descriptive alignment, followed by category-structured reasoning supervision. To support this framework, we construct two complementary datasets and a benchmark: ReasonLite-42M, with open-form, visually verifiable reasoning captions; ReasonPro-16M, with category-specific reasoning supervision; and RCLIP-Bench for diagno","title":"ReasonCLIP-58M: Visually Grounded Commonsense Reasoning Supervision for CLIP","url":"https://arxiv.org/abs/2606.26794","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.26794v1 Announce Type: cross \nAbstract: CLIP and its variants are widely adopted visual backbones in multimodal systems, but their pretraining remains dominated by descriptive image-text alignment. As downstream applications increasingly demand visually grounded commonsense inference and compositional reasoning, it remains unclear whether CLIP-style encoders can support such reasoning without architectural changes. To address this, we present ReasonCLIP-58M, a continual pretraining framework that integrates large-scale reasoning supervision into CLIP-style models through our two-stage strategy, which progressively integrates reasoning signals while preserving descriptive alignment, followed by category-structured reasoning supervision. To support this framework, we construct two complementary datasets and a benchmark: ReasonLite-42M, with open-form, visually verifiable reasoning captions; ReasonPro-16M, with category-specific reasoning supervision; and RCLIP-Bench for diagno","title":"ReasonCLIP-58M: Visually Grounded Commonsense Reasoning Supervision for CLIP","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-26T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.26794"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:bb04fc7b57ad94bddc731848bbfabea4ac1410c897edd367f586959ebbef4c145d32b333b04142601aec1e979b5aa6299f52c83cd1bc1926f067510a70a5a00e","signer":"crovia.substrate","subject":{"observed_at":"2026-06-26T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.26794"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"877c16f81e80000b514257be1e8108b2a4c56462727e7544d36291e9991eae2e","leaf_index":251148,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"7bf73191b049d7270cd2e9bb2795cd5b85ca0ef7fffd6e258dbd3646db31cb3b","side":"right"},{"sibling":"25d14b88fea0e03f88c5e9732b3c867a02a8d21b7967ac17c3ebc13aa25d9a15","side":"right"},{"sibling":"d25712f42ffaa0d1845d5c4abbf38d955838000db4d5ced79b4f5c0e3e883fa7","side":"left"},{"sibling":"069894c0f68f486c0046b70b1a7b33861cfdbf169c4b1c85f33e829eb0fba0e0","side":"left"},{"sibling":"5b47d68b67bb52310274d163ac365d9c251239f13c5d0c263a48b1fc3cf709d3","side":"right"},{"sibling":"8aaa5b928a106f6579b55eed7638d11fdf96308390d7d48464d01d9c5a0bb70e","side":"right"},{"sibling":"2f14dad8e0a8f8bab0ca581ed1b54b6a2b4cc27363a36e17359b2be932eb3fb9","side":"right"},{"sibling":"846fe23f5401fbe8ae1dd175eade4cb955db0a1adde862418a9a781a48f6283c","side":"right"},{"sibling":"d097e1d1d25acb37e74792d46ad69ba853039e2f2b33a092c9ea32a837aeed84","side":"left"},{"sibling":"b6709caadc8510310ee2ec0b66d1058fcad31c91c65cc2bad6f46a693d553580","side":"right"},{"sibling":"a72c3b8804a37d1a9d18e02e6fdb048bc8ea6b0746909cd2de10bdbabc793737","side":"left"},{"sibling":"803703dc2c50a646fa77b57c0056e9a5126611ba4bcde0d6013ccd6b2d44bdbf","side":"right"},{"sibling":"e78f244b1b8df6d5e3fdc6dd76b5c27d4e6fe3b93b8cd61355497b63d7e4cfe8","side":"left"},{"sibling":"6167cb552ed6871fbf0afcf3db01d1017af7d472b136fcbe5404d1df09f41cc1","side":"right"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":251380,"merkle_root":"e042805d07cd8dc777d49695ad78b8d4ec9721df271ff0245d7706773c30b4a5","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260626T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-26T05:37:59Z","sig_algorithm":"ed25519","signature":"c68a6e827804771acd244208495c5e35c6f417307db3c76192f038dedaa5b5f86e019074a3bea353357ff7457924c4f7907832638bb9a4d3d18e7a7f50c6a10b","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_d714e5c7c87c9757e67b43dedd7a5048811e9c0c6d92e25d1b9569cdd3bd0d82"}}