{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1ac3758d6df64a2429ea2421d0b3a7d149f7d9c333369ea680b434390c1910e7","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1ac3758d6df64a2429ea2421d0b3a7d149f7d9c333369ea680b434390c1910e7","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"84561c5fa4d7c68a351f79cd815e7331a394ff24039386de2a076056c99cfcc1","published":"Tue, 19 May 2026 00:00:00 -0400","receipt_hash":"84561c5fa4d7c68a351f79cd815e7331a394ff24039386de2a076056c99cfcc1","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"84561c5fa4d7c68a351f79cd815e7331a394ff24039386de2a076056c99cfcc1","observed_at":"2026-05-19T04:43:36.782648Z","parent_run_hash":"fefa4c726316a95c5dda9fc1ca07a38a811cf7ffa9825b2f09b365abacd9b32d","published":"Tue, 19 May 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2605.18380v1 Announce Type: new \nAbstract: We introduce an extensive qualitative spatial and temporal reasoning (QSTR) benchmark for evaluating large language models (LLMs). We pose questions concerning compositional reasoning (using composition tables, CT), converse relations, and conceptual neighbourhoods (CN) for QSTR calculi, Point Algebra (PA), Allen's Interval Algebra, Interval and Duration (INDU), Region Connection Calculus (RCC-5, RCC-8, and RCC-22), the nine intersection model, cardinal direction calculus, and STAR. The RCC-22 CN is published here for the first time. An extended benchmark systematically varies question presentation including prefix/infix, words/symbols/nonce terms and schematic descriptions for selected calculi. We report results for contemporary frontier models. All models tested perform better than guessing but none can consistently answer all questions correctly. Performance varies sharply by calculus, with PA being the most straightforward, and RCC-2","title":"QSTRBench: a New Benchmark to Evaluate the Ability of Language Models to Reason with Qualitative Spatial and Temporal Calculi","url":"https://arxiv.org/abs/2605.18380","vendor":"arxiv_cs_ai"},"summary":"arXiv:2605.18380v1 Announce Type: new \nAbstract: We introduce an extensive qualitative spatial and temporal reasoning (QSTR) benchmark for evaluating large language models (LLMs). We pose questions concerning compositional reasoning (using composition tables, CT), converse relations, and conceptual neighbourhoods (CN) for QSTR calculi, Point Algebra (PA), Allen's Interval Algebra, Interval and Duration (INDU), Region Connection Calculus (RCC-5, RCC-8, and RCC-22), the nine intersection model, cardinal direction calculus, and STAR. The RCC-22 CN is published here for the first time. An extended benchmark systematically varies question presentation including prefix/infix, words/symbols/nonce terms and schematic descriptions for selected calculi. We report results for contemporary frontier models. All models tested perform better than guessing but none can consistently answer all questions correctly. Performance varies sharply by calculus, with PA being the most straightforward, and RCC-2","title":"QSTRBench: a New Benchmark to Evaluate the Ability of Language Models to Reason with Qualitative Spatial and Temporal Calculi","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-05-19T04:43:36Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2605.18380"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a75be148dab430622bc9fd9a51a0bc0aadbc533e54ca7590808f660dc6626babc25004af154a1994bb2ca1866f957155008c98f62510ffbe43acaa56c8385f03","signer":"crovia.substrate","subject":{"observed_at":"2026-05-19T04:43:36Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2605.18380"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"200ead11a2368ebc26d131915ce948f39baa9bfe2ff65b514c19a7a8e026b542","leaf_index":142474,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"a7f2e5e43591c0d0e6e585ebfe3a97666bcf640c27924eb9f3af34a7218f0391","side":"right"},{"sibling":"88d10e0bd988c1c9afc9703dab6c108b6d61e1305d863732d3349d6a5601ebb9","side":"left"},{"sibling":"aec3457f4a0f540638ad273f0ae3fd7f34cfc0838682f6389634eb198ce75bdf","side":"right"},{"sibling":"576a2f3115c9f77078347d6b2f93194a95ef9913cad61c6fc378e67e873a061e","side":"left"},{"sibling":"9e43ff9e65458c0a6b2fcaf63507a0e7dbe7b0b0a1658e5b8cec3d36e6c62142","side":"right"},{"sibling":"814692ef8f42fc874bcbe7d0b13008b29149d148c375d39826913ef2ee30e4ca","side":"right"},{"sibling":"a77aa4ca4c98cdb1d03d5011f1cffd38c870ac96455c8855027846c6ace8c288","side":"right"},{"sibling":"fb0c5016a320d9b8a5cadece4783693f5f57854143a494aef80e914b671404d1","side":"left"},{"sibling":"bbd9a0327a91b5df2093631f9bf2665ba64be214117a6f2060ca0e6e4c6bf27d","side":"right"},{"sibling":"2e0ce989d789c88e796991ef014ce7e4e1f96c0d4ede9da8d31afcf5ba6a8f46","side":"right"},{"sibling":"202f1bead178ef3785968d50d3d188264a95192a077654c331612e04a34cbfbe","side":"left"},{"sibling":"72249c8c8b068386e35d16f4bd0bbeb9ba820ca217ef0f0d28396c9fe493f5f0","side":"left"},{"sibling":"ea64599340f7ffdf17ad0cbc1d9401ef8870a347e3847bdc106d06b1673df09c","side":"right"},{"sibling":"8f4c0fbe56b6c010fbb8c782ebcd478079bb3f991d8704e2534209a075d9163c","side":"left"},{"sibling":"4db1f363729507e27a60851cf6ed334d7b9acdef194ed7d419aba4d2bd367a4a","side":"right"},{"sibling":"a86ee18c45e7fcc408b6007eaece05aa75b2d9ae30252e9e878462b4dffbef7b","side":"right"},{"sibling":"1d18e7663d43ccff0122ecc7ee12645bb16afb607b218e81b1ea2408f863cb78","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":143302,"merkle_root":"999156d40a7c61d9ddd52b7338f3cbda3e68f53bace070c7b616ea194e23b123","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260519T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-05-19T05:37:30Z","sig_algorithm":"ed25519","signature":"b1a252cc66ff32bed1d10dd88a6b2a200e3856d3dbcfcc4ee55e02e00f3d548e854ed9c544704b222bd5d315492c4a935ba2d90d727c585a67899b0ad602fc05","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1ac3758d6df64a2429ea2421d0b3a7d149f7d9c333369ea680b434390c1910e7"}}