{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_390737a62d360e6441346454402ef6d4c4b14280daa42d2e9e22bb587d2a156f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_390737a62d360e6441346454402ef6d4c4b14280daa42d2e9e22bb587d2a156f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"d1aae2ed3eb5a9da95bc8c74ef99bd5573a1a1379de99394a0c8ad4f5ddd4a9b","published":"Mon, 13 Jul 2026 00:00:00 -0400","receipt_hash":"d1aae2ed3eb5a9da95bc8c74ef99bd5573a1a1379de99394a0c8ad4f5ddd4a9b","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"d1aae2ed3eb5a9da95bc8c74ef99bd5573a1a1379de99394a0c8ad4f5ddd4a9b","observed_at":"2026-07-13T04:43:08.394955Z","parent_run_hash":"900c1c934245e788564e199a9619f2dc36ec91d9804ddd9c6a40fb42c8a1e1c0","published":"Mon, 13 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.09438v1 Announce Type: cross \nAbstract: Test-time scaling (TTS) reliably improves reasoning in large language models, but whether it transfers to small open vision-language models remains unclear. We examine this on EXAMS-V, a multilingual visual multiple-choice benchmark, comparing self-consistency, describe-then-reason with PRM-guided beam search, and two post-hoc selectors across Qwen2.5-VL-7B-Instruct and Qwen3.5-4B. What matters is the conditions under which TTS runs, not the search or verification machinery. The largest factor is parseability: an early prompt format left many chains reasoning correctly yet never committing to an answer letter, which a standard answer cue and a guided repair step largely remove. A larger decoding budget removes the rest: raising the per-chain token limit from 1k to 2k recovers 3.7 pp, whereas sampling more chains (8 to 16) adds only 0.15 pp. Once chains have room to finish, elaborate methods contribute little: PRM-guided beam search tra","title":"Test-Time Scaling for Small VLMs on Multilingual Visual MCQ","url":"https://arxiv.org/abs/2607.09438","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.09438v1 Announce Type: cross \nAbstract: Test-time scaling (TTS) reliably improves reasoning in large language models, but whether it transfers to small open vision-language models remains unclear. We examine this on EXAMS-V, a multilingual visual multiple-choice benchmark, comparing self-consistency, describe-then-reason with PRM-guided beam search, and two post-hoc selectors across Qwen2.5-VL-7B-Instruct and Qwen3.5-4B. What matters is the conditions under which TTS runs, not the search or verification machinery. The largest factor is parseability: an early prompt format left many chains reasoning correctly yet never committing to an answer letter, which a standard answer cue and a guided repair step largely remove. A larger decoding budget removes the rest: raising the per-chain token limit from 1k to 2k recovers 3.7 pp, whereas sampling more chains (8 to 16) adds only 0.15 pp. Once chains have room to finish, elaborate methods contribute little: PRM-guided beam search tra","title":"Test-Time Scaling for Small VLMs on Multilingual Visual MCQ","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-13T04:43:08Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.09438"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a29ff0af0fd8f353d5cca0a1c69032168ec92288ef620a70bb805d544b9a7751b45cf68c29f6078eada0a1da04a3ee6ff84f3737bb9a37c64cbe9308d3355a0f","signer":"crovia.substrate","subject":{"observed_at":"2026-07-13T04:43:08Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.09438"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"7e35f32027befef2d293eedf6865d482df5ca36c895c6c8c31295304337b2752","leaf_index":309665,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"bee6bb04fa0e977577200c9054cbd050b80335a48bc8dec82f2867e4dc2e3474","side":"left"},{"sibling":"5c9d44cf0c53c76aa8a949571c72bbc53aff6f693bf6707e38bebbc3291f6b14","side":"right"},{"sibling":"175c2239cd194a6e1dbcadb36239455e85625b61219c8a602afd58db1ede6313","side":"right"},{"sibling":"cb1cfbda5600884ac1e6a4bc0d2cea36dece7c230a0d124d00d850cefe049f4e","side":"right"},{"sibling":"ade4794d173c42e632d7e185c354b26ce0b5511a29667e60a600752aa4e1b258","side":"right"},{"sibling":"817e14c6ab3a66a77efac57ff4cc63efe5efe647afce47f5aefa03dc77fc1f24","side":"left"},{"sibling":"a06f2db4ff89b8806c463205a6d957aa0950263fff991c2aa1386688c611457f","side":"right"},{"sibling":"49f7764de5dea797b32547b8437f6d371cde9fb33ac6cf784651dcfa196b149a","side":"left"},{"sibling":"6cbb0c4695e74fa8ec17dfde89fa56c1958a1d4dffbbcbe3556da4a69c699ebc","side":"left"},{"sibling":"c4bde3283be97905033d39c1097ac73b82f180de8830384fe6f9f1933f7fa4b9","side":"right"},{"sibling":"3b5f968ebea87e7be458ef7e63c6637988d27ba6d170e1f3794d385cd76ca23f","side":"right"},{"sibling":"5ed534e945b31085c140b50460415e7960b76a4b6266da67d272b853dc94b352","side":"left"},{"sibling":"91010b271bc8eb5253b3549292ef3146e1d85bc9bac7ca36e0862f1204f84e8d","side":"left"},{"sibling":"772fbc112e94d8e574379343387c65503d3cf8fb16ff89090174eccced871a41","side":"left"},{"sibling":"5d50450cae1a230f682b390c8e28ae822ec6c0c04a9bc79b0d27af97306ccccd","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"e9ac4b1e71d3f572751b7984e9ca00893d71628137b4634e82df02cf3db9ab68","side":"right"},{"sibling":"99ff86058e045249bf936a629308be31e4cc71328f0282c4a924f4e6718be5f0","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":309862,"merkle_root":"f18a76abb66e7cb448986b6541091416ed4ebdde8124f5208a7c7c94bb4165d1","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260713T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-13T05:38:23Z","sig_algorithm":"ed25519","signature":"0488e3527b1d559ba55116220f91e2f6c22bb5358a135e8a9b94a65450b50511702044fa993555662ebca09f005b277f5f6ad0d044ec96a5568e9993c09ef007","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_390737a62d360e6441346454402ef6d4c4b14280daa42d2e9e22bb587d2a156f"}}