{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b1b47032f99c7da25d02ac270fc163182f649bf5ffd640ea892554d861fb121e","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b1b47032f99c7da25d02ac270fc163182f649bf5ffd640ea892554d861fb121e","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6300c1c63a3f23e9704aa96b7592dafcfb2d3f698199a7061430ee5b83149d44","published":"Wed, 10 Jun 2026 00:00:00 -0400","receipt_hash":"6300c1c63a3f23e9704aa96b7592dafcfb2d3f698199a7061430ee5b83149d44","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6300c1c63a3f23e9704aa96b7592dafcfb2d3f698199a7061430ee5b83149d44","observed_at":"2026-06-10T04:43:37.461885Z","parent_run_hash":"23aff1a6f676ba7ca33f70f4ddfae1dd282fb86104d577ce9be510d81a94c5dc","published":"Wed, 10 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.11070v1 Announce Type: cross \nAbstract: Recent advances in reasoning and tool-calling capabilities of large language models (LLMs) have enabled increasingly capable agentic systems. However, existing benchmarks remain limited in task complexity, realism, and domain diversity, and often fail to capture interactions that span multiple domains, limiting their ability to evaluate agents in realistic multi-step settings that require sustained reasoning and coordination. To address these limitations, we introduce T1-Bench, a high-fidelity, comprehensive benchmark for evaluating agentic systems in realistic customer-facing, multi-domain environments, featuring interleaved scenarios that require structured reasoning across multi-turn user-assistant interactions and substantially increasing both compositional complexity and evaluative rigor across 25 domains of varying difficulty. We evaluate T1-Bench using 12 proprietary and open-weight models, providing a reproducible and standardi","title":"T1-Bench: Benchmarking Multi-Scenario Agents in Real-World Domains","url":"https://arxiv.org/abs/2606.11070","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.11070v1 Announce Type: cross \nAbstract: Recent advances in reasoning and tool-calling capabilities of large language models (LLMs) have enabled increasingly capable agentic systems. However, existing benchmarks remain limited in task complexity, realism, and domain diversity, and often fail to capture interactions that span multiple domains, limiting their ability to evaluate agents in realistic multi-step settings that require sustained reasoning and coordination. To address these limitations, we introduce T1-Bench, a high-fidelity, comprehensive benchmark for evaluating agentic systems in realistic customer-facing, multi-domain environments, featuring interleaved scenarios that require structured reasoning across multi-turn user-assistant interactions and substantially increasing both compositional complexity and evaluative rigor across 25 domains of varying difficulty. We evaluate T1-Bench using 12 proprietary and open-weight models, providing a reproducible and standardi","title":"T1-Bench: Benchmarking Multi-Scenario Agents in Real-World Domains","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-10T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.11070"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:7095044f7116934c37fc01764f2b7e2cdd201e694047be9c31340dcdd03df85a7d643c345b38af4f51147586707c1f9e82054bf57a7b178e0e0f18b668943e09","signer":"crovia.substrate","subject":{"observed_at":"2026-06-10T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.11070"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"b9898f4da6281267c710276cf1626bc218c42060c9d2e9d875b25ab411d8fee3","leaf_index":226108,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"93637dc2968bb375b392ecc62d252bebe9e78ab54beab437b9830bcaf157cb99","side":"right"},{"sibling":"759e979b989391469ac6ec5e94f1f2dc6e2e6548ded0ad146bdb7c63073b4384","side":"right"},{"sibling":"f03d1e081818128819ac367595e0652cac35943139338435a6a02488121b7960","side":"left"},{"sibling":"c959c6d7deee4d8bcac914d6aa9c120b997f4511d83640524c5e0a3af27308e2","side":"left"},{"sibling":"83c53580f3270a0dd44187601784fff31f2c5b74f40c703fc8f55298b3cae0af","side":"left"},{"sibling":"c45d6316d9967be11fb5ebc30788b1e275d0bcba0f31abb481c71675b76a224d","side":"left"},{"sibling":"88b2be21cb27d1f77926cea4c04fb2bafb96cdf167c1655dc8560cea37cb31bf","side":"right"},{"sibling":"670eed753a274b9b7494adeace1b88e7adfaecd7eeaa78b82e0fc83a75d20865","side":"right"},{"sibling":"b180cfc3f8638c912e90139ec42e2e3fd9b67b3fe342b36e8954314633ae5f61","side":"left"},{"sibling":"a4d17aefe58175050dc159af6246658fcf1c9f3ed57aacf1b350fc3261de4e69","side":"left"},{"sibling":"280b980aa0c7756b0b0cb22658f26466d36f0e70fbc3312cd2311d9898e30b8f","side":"right"},{"sibling":"c98954d4b658b1dda60fe52576fcf9bf21a2d49c67fb63f8c30f16ab5f721938","side":"right"},{"sibling":"cdb58f86163046d3b15f857b03372ec75e1ad9ea4548e086793d528b9eed364d","side":"left"},{"sibling":"533d82482604463aa4a281b9d8b85917b383b7c5f924b2b494039524c55e8797","side":"left"},{"sibling":"b2590791b920ca2a4ed39de126d2c0b1a10d9e7e62f572f12425f214e767b6e1","side":"left"},{"sibling":"6cea4964f32722eb370847c2f7c9d6a9f0622c239538b07e6815a59d6fd8d49c","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":228173,"merkle_root":"7e416202c0bfd759bd2eea4236713b403993d99793fe8badb5065040080bece3","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260611T143708Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-11T21:59:35Z","sig_algorithm":"ed25519","signature":"231c80024bc3982dd493c45b31af95097e97aabc6d712a4e5bad7d0cbdd3c08e01ff395b0f8e72754bac97016e0cd0eed88b8a13cb71edbbcb9b6d72c10a7b03","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b1b47032f99c7da25d02ac270fc163182f649bf5ffd640ea892554d861fb121e"}}