{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_553384ea2bd2fe12a73b97213c5b89553a244f8406ceb348b107c9c66aa5df86","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_553384ea2bd2fe12a73b97213c5b89553a244f8406ceb348b107c9c66aa5df86","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"50551b36c81a45a3bceb165ef3c6ac58d7fafd4d44cb954f5ccaa5ac490d36b1","published":"Wed, 17 Jun 2026 00:00:00 -0400","receipt_hash":"50551b36c81a45a3bceb165ef3c6ac58d7fafd4d44cb954f5ccaa5ac490d36b1","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"50551b36c81a45a3bceb165ef3c6ac58d7fafd4d44cb954f5ccaa5ac490d36b1","observed_at":"2026-06-17T04:43:17.968423Z","parent_run_hash":"8f56c4deb22b28178ba7974d6ffc5ff17336d44c95dd94de80095705245fa113","published":"Wed, 17 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.17678v1 Announce Type: cross \nAbstract: Multimodal large language models (MLLMs) integrate strong text reasoning with visual inputs, yet their responses can be inconsistent with the underlying images, indicating ineffective utilization of visual evidence during inference. The prevailing training paradigm relies on large-scale caption-based pretraining for general alignment, followed by supervised fine-tuning and reinforcement learning to enable instruction following and complex reasoning. However, such pretraining provides only weak visual grounding: short, coarse captions bias models toward salient objects while neglecting fine-grained visual evidence. In this paper, we introduce Visual Evidence Pre-Alignment (VEPA), an intermediate stage between pretraining and post-training that explores a novel sufficiency-driven objective with Group Relative Policy Optimization (GRPO) to optimize question-conditioned visual evidence descriptions. Extensive experiments across diverse ben","title":"See First, Answer Later: Visual Evidence Pre-Alignment via Sufficiency-Driven RL","url":"https://arxiv.org/abs/2606.17678","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.17678v1 Announce Type: cross \nAbstract: Multimodal large language models (MLLMs) integrate strong text reasoning with visual inputs, yet their responses can be inconsistent with the underlying images, indicating ineffective utilization of visual evidence during inference. The prevailing training paradigm relies on large-scale caption-based pretraining for general alignment, followed by supervised fine-tuning and reinforcement learning to enable instruction following and complex reasoning. However, such pretraining provides only weak visual grounding: short, coarse captions bias models toward salient objects while neglecting fine-grained visual evidence. In this paper, we introduce Visual Evidence Pre-Alignment (VEPA), an intermediate stage between pretraining and post-training that explores a novel sufficiency-driven objective with Group Relative Policy Optimization (GRPO) to optimize question-conditioned visual evidence descriptions. Extensive experiments across diverse ben","title":"See First, Answer Later: Visual Evidence Pre-Alignment via Sufficiency-Driven RL","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-17T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.17678"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ec23cdf044dff3e0c68ab026a24ece18254200a549097acde303d1bc1e541575a82425fe4cb226fabc5594509cdf7d34f784f03eb63cc15dec3aa2b1ed2b840c","signer":"crovia.substrate","subject":{"observed_at":"2026-06-17T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.17678"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f02fdceacd5594e1c05bf4b329a1036a62b99e1a2cff9e4c19c980c97748f71d","leaf_index":230995,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"aed97d6caf13536af660db88f72144c07c014d78f725c465a7df2d8854b569ff","side":"left"},{"sibling":"3c22d68fbc9903a3fb89358eb7b4cbea6355f2e83fff9dcb2306ac3a208c535c","side":"left"},{"sibling":"525e1565bf03cda7221d8c1cbca17868c5b350820a56601a323d8082e416ca1f","side":"right"},{"sibling":"877c8f892abbe1ab88d2e30ad3aaee8ae3f32c06670b543160be3363a1a8fb7a","side":"right"},{"sibling":"5e77d6396e1cfd34095b5670f9fff36ce3601e2b5fb5645bd9d8e62e9f93b3d2","side":"left"},{"sibling":"b81e03292fa07ff830b444c07ef37ec24747d2035531f28f436dbc6110f3590d","side":"right"},{"sibling":"8daceae0255c786ffdef12a608da0c1a09b9a2ceedfcc067120bfd86b4ab9952","side":"left"},{"sibling":"ce828166a6a4fc2ff2681c985a56053a1f8b683cc245e0b232fa5939310b3ad9","side":"right"},{"sibling":"ebbec9ce4bf43a3f5f71e2c07df4c91301abefa2da727a4d748099a74e980bc3","side":"right"},{"sibling":"1d74941c32baeab8cad08f8700cb49427d8256f231534f2a225b2bb3e84e4ff8","side":"left"},{"sibling":"d5b9f8b1a2c9f6a46e17982dfbe6ce1f3b5fa4e730220397f2253d114dcc8486","side":"left"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_553384ea2bd2fe12a73b97213c5b89553a244f8406ceb348b107c9c66aa5df86"}}