{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_8ea2734db68fef44f0d6053c29bdd363a377af730f17dee3187a77a66e8bb242","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_8ea2734db68fef44f0d6053c29bdd363a377af730f17dee3187a77a66e8bb242","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"17120aedaa4df199dea2efc68be2893e35ec5441f3a71e59678ad94c5b70a511","published":"Mon, 01 Jun 2026 00:00:00 -0400","receipt_hash":"17120aedaa4df199dea2efc68be2893e35ec5441f3a71e59678ad94c5b70a511","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"17120aedaa4df199dea2efc68be2893e35ec5441f3a71e59678ad94c5b70a511","observed_at":"2026-06-01T04:43:13.859018Z","parent_run_hash":"8993bbc535dae8c9669e099af3624cb39166b8d9bbfd66f26ae5c338cbb21be2","published":"Mon, 01 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2602.13812v3 Announce Type: replace-cross \nAbstract: Document-to-table (Doc2Table) extraction derives structured tables from unstructured documents under a target schema, enabling reliable and verifiable SQL-based data analytics. Although large language models (LLMs) have shown promise in flexible information extraction, their ability to produce precisely structured tables remains insufficiently understood, particularly for indirect extraction that requires complex capabilities such as reasoning and conflict resolution. Existing benchmarks neither explicitly distinguish nor comprehensively cover the diverse capabilities required in Doc2Table extraction. We argue that a capability-aware benchmark is essential for systematic evaluation. However, constructing such benchmarks using human-annotated document-table pairs is costly, difficult to scale, and limited in capability coverage. To address this, we adopt a reverse Table2Doc paradigm and design a multi-agent synthesis workflow to","title":"DTBench: A Synthetic Benchmark for Document-to-Table Extraction","url":"https://arxiv.org/abs/2602.13812","vendor":"arxiv_cs_ai"},"summary":"arXiv:2602.13812v3 Announce Type: replace-cross \nAbstract: Document-to-table (Doc2Table) extraction derives structured tables from unstructured documents under a target schema, enabling reliable and verifiable SQL-based data analytics. Although large language models (LLMs) have shown promise in flexible information extraction, their ability to produce precisely structured tables remains insufficiently understood, particularly for indirect extraction that requires complex capabilities such as reasoning and conflict resolution. Existing benchmarks neither explicitly distinguish nor comprehensively cover the diverse capabilities required in Doc2Table extraction. We argue that a capability-aware benchmark is essential for systematic evaluation. However, constructing such benchmarks using human-annotated document-table pairs is costly, difficult to scale, and limited in capability coverage. To address this, we adopt a reverse Table2Doc paradigm and design a multi-agent synthesis workflow to","title":"DTBench: A Synthetic Benchmark for Document-to-Table Extraction","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-01T04:43:13Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2602.13812"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:5087b234ce4e5b41aa14b532b8fd30b4c1717bc7d033b7b156891b8328c1d8b5fb4bd3a9865d744d7596b07679ba2f5ae1724f825e999a44a61bb5247cbae60f","signer":"crovia.substrate","subject":{"observed_at":"2026-06-01T04:43:13Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2602.13812"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"8ea11cbf1cba1e39b4512c5bba89919faeb01434584e19efc0eedc7b94535be0","leaf_index":164107,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"fb131438db39d0e6fb15a48c933cd93626e982e20ae2a26c257f9d043087c27c","side":"left"},{"sibling":"28486e1b4d3475846be5d3a60a988a45d4dd01f1fe0e26e5ea4679ded286c43b","side":"left"},{"sibling":"fa83190f6d7510b32e91aafad3c2323f98d4c7620ddf853b153477f89b3f4525","side":"right"},{"sibling":"d92d07fc32df58040506344d9e79b91720691adba3c62f5fee5383649784e3a1","side":"left"},{"sibling":"d2ed68514682bda888bd26f500bf818b9f1c0e894fd25ac3c568cf7a35bee4ae","side":"right"},{"sibling":"db716d28ae117a529404e8f1c8ea86faefa948493e597acb82ec3b14ff0636f4","side":"right"},{"sibling":"411a66cd5a7fa8f996afb73d603026ea06f5f72e02002d390e8c35173a12a73e","side":"right"},{"sibling":"fa66353dd6f42c595db53c4b87c558d4cbcc04862c4058c0562cc7d5d9015f15","side":"right"},{"sibling":"7caaac9d7d329e12cd7433a2a627fcb2eaa744feb44b4a8462db35186e31e111","side":"left"},{"sibling":"fc601f0745c37fc5f7c249300e654db05d61bb9059885eeb0b0a5047b5d28408","side":"right"},{"sibling":"6ed9290cdae063f13bfeb71c4c5440595cabb61fd4b39225b9cc913ffb336dd7","side":"right"},{"sibling":"a0446b923d1ce90021e78edff07f6bfc7cc2a1326a565c5f0786b82a24dd0a2a","side":"right"},{"sibling":"e598fd53912c30e58ca8e58d7d8a338fe0f2ecb63fdd99225bc703c499c948ec","side":"right"},{"sibling":"fc4873333221ec8167697f75b6f6a8a08491a8cf18952defb65fb6d4958fa5e7","side":"right"},{"sibling":"05c8a827da2a05549ee3250310777009120c687885816bf6c7c74801bfaa346d","side":"right"},{"sibling":"5ea2f2dc9f046df723b6bd9932d61a9a3d80a76e79ce1b939b3d93ff79b5a91c","side":"left"},{"sibling":"ce41d9b82f34b16efd653dfb3552acc4e2512939e47903e5fc979fbed00c5764","side":"right"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":164217,"merkle_root":"a1098816aea1b60b8fe37b62410469bc5024a2c335bbec4f6ef2add7875dbdf2","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260601T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-01T05:37:41Z","sig_algorithm":"ed25519","signature":"d7f91db1d54b9495c499440c2828f4bd53360555391ce6e25adea5183bc1fa0f697d80708a099d0b0429e6f8cb6c71e7fccf82acb3c84481149974fb26074708","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_8ea2734db68fef44f0d6053c29bdd363a377af730f17dee3187a77a66e8bb242"}}