{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_26d526de69f153c5e18e65517e6c38e48e5a33e75e30eeb72a5f41151402c410","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_26d526de69f153c5e18e65517e6c38e48e5a33e75e30eeb72a5f41151402c410","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"250b0397a717658b423b7d771a9440315a12c8e9660f55ec56797b86f9d7fe00","published":"Tue, 19 May 2026 00:00:00 -0400","receipt_hash":"250b0397a717658b423b7d771a9440315a12c8e9660f55ec56797b86f9d7fe00","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"250b0397a717658b423b7d771a9440315a12c8e9660f55ec56797b86f9d7fe00","observed_at":"2026-05-19T04:43:36.782648Z","parent_run_hash":"fefa4c726316a95c5dda9fc1ca07a38a811cf7ffa9825b2f09b365abacd9b32d","published":"Tue, 19 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.17278v1 Announce Type: new \nAbstract: Abstract reasoning ability reflects the intelligence and generalization capacity of LLMs to extract and apply abstract rules. However, accurately measuring this ability remains challenging: existing benchmarks either rely on expensive manual annotation, limiting their scale, or risk measuring memorization rather than genuine reasoning. To address this, we introduce an automated pipeline named A2RBench, encompassing generation, expansion, evaluation, and analysis. Specifically, in the generation stage, LLMs create diverse tasks demanding genuine reasoning; in the expansion stage, LLMs reuse validated rules and expand new input spaces to generate task variations, achieving scaling. However, such a process may cause hallucinations. To eliminate it, we further establish a theoretical framework and prove that programmatic verification--testing whether the inverse operation perfectly reverses the forward operation (cycle consistency)--guarante","title":"A2RBench: An Automatic Paradigm for Formally Verifiable Abstract Reasoning Benchmark Generation","url":"https://arxiv.org/abs/2605.17278","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.17278v1 Announce Type: new \nAbstract: Abstract reasoning ability reflects the intelligence and generalization capacity of LLMs to extract and apply abstract rules. However, accurately measuring this ability remains challenging: existing benchmarks either rely on expensive manual annotation, limiting their scale, or risk measuring memorization rather than genuine reasoning. To address this, we introduce an automated pipeline named A2RBench, encompassing generation, expansion, evaluation, and analysis. Specifically, in the generation stage, LLMs create diverse tasks demanding genuine reasoning; in the expansion stage, LLMs reuse validated rules and expand new input spaces to generate task variations, achieving scaling. However, such a process may cause hallucinations. To eliminate it, we further establish a theoretical framework and prove that programmatic verification--testing whether the inverse operation perfectly reverses the forward operation (cycle consistency)--guarante","title":"A2RBench: An Automatic Paradigm for Formally Verifiable Abstract Reasoning Benchmark Generation","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-19T04:43:36Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.17278"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:17801880efe8b1c76e4330cfd905b0a344e6e531f6b24bd442b947f2a9db05b6893e11b3b8d157ed81860dfee17c198a4f3940f9cebe0898116596dad2093203","signer":"crovia.substrate","subject":{"observed_at":"2026-05-19T04:43:36Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.17278"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"9fba6a9136d378a5dbac2e62d1e9d2f0d6dfe2267c69cc49358bb10a2f6e66b5","leaf_index":142405,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"de79300552d79544cc1dd51d405dfe7bf1bb215df2feaadd3b1d28c169b72605","side":"left"},{"sibling":"8688eb2d7cd70663c67c5f9a545ddf08cdd1ecbb6c374ce679bb3dd32288f16d","side":"right"},{"sibling":"05cebcaec877b6dcca71bee179ab638f73d04e1ff1ca5bd5a0ffb10fd05f2ff0","side":"left"},{"sibling":"cc8172b7b795ff7325854445a665f60a66dfc98da4205330141821fc4d9624c7","side":"right"},{"sibling":"60d907601ad12446188e031e90d99b6d1c6cc6fba57dd0c9757078df2544129c","side":"right"},{"sibling":"fad41496adcbc679335728a8a78d579e89155a19c51950d4a6696c02c4793709","side":"right"},{"sibling":"d0b56e9dc101b8d881b8d3739227c4d0aaca1cf96fef1945825f5d56fd5664ff","side":"left"},{"sibling":"6b37a038a508ab9f6e9513e2a88094021d9ee3a5e516d7a811c9cd8e3ad55e3c","side":"right"},{"sibling":"bbd9a0327a91b5df2093631f9bf2665ba64be214117a6f2060ca0e6e4c6bf27d","side":"right"},{"sibling":"2e0ce989d789c88e796991ef014ce7e4e1f96c0d4ede9da8d31afcf5ba6a8f46","side":"right"},{"sibling":"202f1bead178ef3785968d50d3d188264a95192a077654c331612e04a34cbfbe","side":"left"},{"sibling":"72249c8c8b068386e35d16f4bd0bbeb9ba820ca217ef0f0d28396c9fe493f5f0","side":"left"},{"sibling":"ea64599340f7ffdf17ad0cbc1d9401ef8870a347e3847bdc106d06b1673df09c","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"4db1f363729507e27a60851cf6ed334d7b9acdef194ed7d419aba4d2bd367a4a","side":"right"},{"sibling":"a86ee18c45e7fcc408b6007eaece05aa75b2d9ae30252e9e878462b4dffbef7b","side":"right"},{"sibling":"1d18e7663d43ccff0122ecc7ee12645bb16afb607b218e81b1ea2408f863cb78","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":143302,"merkle_root":"999156d40a7c61d9ddd52b7338f3cbda3e68f53bace070c7b616ea194e23b123","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260519T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-19T05:37:30Z","sig_algorithm":"ed25519","signature":"b1a252cc66ff32bed1d10dd88a6b2a200e3856d3dbcfcc4ee55e02e00f3d548e854ed9c544704b222bd5d315492c4a935ba2d90d727c585a67899b0ad602fc05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_26d526de69f153c5e18e65517e6c38e48e5a33e75e30eeb72a5f41151402c410"}}