{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_bb1e544b4675e7fe35d2a48985e17bbd37fc75a0f70a22e97d340cb374f7e90d","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_bb1e544b4675e7fe35d2a48985e17bbd37fc75a0f70a22e97d340cb374f7e90d","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"f8f128ffce5a9686fe28f43758e3689674ead4eadda3a8c9561873e99aa45afd","published":"Thu, 18 Jun 2026 00:00:00 -0400","receipt_hash":"f8f128ffce5a9686fe28f43758e3689674ead4eadda3a8c9561873e99aa45afd","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"f8f128ffce5a9686fe28f43758e3689674ead4eadda3a8c9561873e99aa45afd","observed_at":"2026-06-18T04:43:37.219665Z","parent_run_hash":"de79a40f7b3537d88842f7ac355e799c5df2adcb4fc32a4e28096d4bbdf01739","published":"Thu, 18 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2601.00567v2 Announce Type: replace-cross \nAbstract: Adapting general-domain retrievers to scientific domains is challenging due to the scarcity of large-scale domain-specific relevance annotations and the substantial mismatch in vocabulary and information needs. Recent approaches address these issues through two independent directions that leverage large language models (LLMs): (1) generating synthetic queries for fine-tuning, and (2) generating auxiliary contexts to support relevance matching. However, both directions overlook the diverse academic concepts embedded within scientific documents, often producing redundant or conceptually narrow queries and contexts. To address this limitation, we introduce an academic concept index, which extracts key concepts from papers and organizes them guided by an academic taxonomy. This structured index serves as a foundation for improving both directions. First, we enhance the synthetic query generation with concept coverage-based generati","title":"Improving Scientific Document Retrieval with Academic Concept Index","url":"https://arxiv.org/abs/2601.00567","vendor":"arxiv_cs_ai"},"summary":"arXiv:2601.00567v2 Announce Type: replace-cross \nAbstract: Adapting general-domain retrievers to scientific domains is challenging due to the scarcity of large-scale domain-specific relevance annotations and the substantial mismatch in vocabulary and information needs. Recent approaches address these issues through two independent directions that leverage large language models (LLMs): (1) generating synthetic queries for fine-tuning, and (2) generating auxiliary contexts to support relevance matching. However, both directions overlook the diverse academic concepts embedded within scientific documents, often producing redundant or conceptually narrow queries and contexts. To address this limitation, we introduce an academic concept index, which extracts key concepts from papers and organizes them guided by an academic taxonomy. This structured index serves as a foundation for improving both directions. First, we enhance the synthetic query generation with concept coverage-based generati","title":"Improving Scientific Document Retrieval with Academic Concept Index","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-18T04:43:37Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2601.00567"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:af2aa8224c18f1627edb9fc8c4f07b43f3c872a7ee855062425c0633a09a06ec11fa54e8a0737b0f3221f6667582355d00a96eb47fa855c0231662d8f3359009","signer":"crovia.substrate","subject":{"observed_at":"2026-06-18T04:43:37Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2601.00567"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"f343092aa72c9845928e1e77db882dbc847afad6bd0d1e530eb651345fac6d20","leaf_index":233481,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"f88d791e6b1f42fd7840c9a06514efd60ac9a5f34afc62650239e7fd6c51c1e1","side":"left"},{"sibling":"b5d100c73d913e9b77f5a63cd866399e24a7eebf9c61d1a44681f380c62a4dc7","side":"right"},{"sibling":"2ad8a28095035d2d5d30ddd25a56aedabb7386faa115ffa5f8a20c6028754120","side":"right"},{"sibling":"8ee80e90a5859a0f3970010c25af4705db79811ae9dd29e704d29144f4182b75","side":"left"},{"sibling":"7649f67f4250c1762457d97675664e11db17378d8f2e1217749de00c26d834bc","side":"right"},{"sibling":"8ea5d1acc478cab779b8d8c6bd66dc15ad77122446a534f70804af03d690458e","side":"right"},{"sibling":"5e436eec5370aaf4cccb6c432bd004f0a554eab0e19ec87f70ba66470ad7303b","side":"right"},{"sibling":"fe9c643bdeb268143f61f15d89d0ff9dd03becb2914656c7a873d95f7266b5ce","side":"right"},{"sibling":"ad9e1cf26277141407aebd8692bfb16135dd1e614913e54b1dd445fed15d105a","side":"right"},{"sibling":"b151db1a7de0ce9a329250fae8b690f5b55ab3cfe5468a1fbb5f5bde0703b420","side":"right"},{"sibling":"7dc9143c057343b46a3b988492fba5262dec443cc1fb6070e3ea81543ca6a522","side":"right"},{"sibling":"ad5850946feb9a22361b5b9df0884f9ef1edcc7efea0374ff5782080ccb1a947","side":"right"},{"sibling":"410c633928fea11c5b4bdddb431956b1d7c320db9cda00d2fe32e0fcf888d7b7","side":"left"},{"sibling":"e616c34dbaf9456d5a6d3e2da82cde8621293c9f6d8a4cf9e441d7fd9cc81579","side":"right"},{"sibling":"94c0c932e61657f5e37fdba43f6ca9eddea8359425a7c1558dabe566911d5304","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":234491,"merkle_root":"02576a6980e38bab47864ae2c57b5a5ff21e554e9bdf8f64bdf28155ff1aabec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260618T143732Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-18T18:33:39Z","sig_algorithm":"ed25519","signature":"b6c708778fc38b7789a2b91156cfe87252a7cd3a1d29121cba11a0c78f8cf104ca3019fc50a962fa5a216bcc4922fc8f3f69c04d1f8332c6dc0d931e1012e502","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_bb1e544b4675e7fe35d2a48985e17bbd37fc75a0f70a22e97d340cb374f7e90d"}}