{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_a7717822c7d16502e9785d3eeb267386bb7aa3d5803a1f09b2d8d100b1e6d968","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_a7717822c7d16502e9785d3eeb267386bb7aa3d5803a1f09b2d8d100b1e6d968","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"a979c2e669f22161625dd85516648ae7c5082bca9d976a9fbf27421badc3f7e2","published":"Wed, 17 Jun 2026 00:00:00 -0400","receipt_hash":"a979c2e669f22161625dd85516648ae7c5082bca9d976a9fbf27421badc3f7e2","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"a979c2e669f22161625dd85516648ae7c5082bca9d976a9fbf27421badc3f7e2","observed_at":"2026-06-17T04:43:17.968423Z","parent_run_hash":"8f56c4deb22b28178ba7974d6ffc5ff17336d44c95dd94de80095705245fa113","published":"Wed, 17 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.17588v1 Announce Type: cross \nAbstract: Several studies have examined the use of large language models (LLMs) for title-abstract screening in systematic reviews (SRs), reporting mixed accuracy. However, questions of reliability remain largely unaddressed. In this study, we go beyond quantitative LLM-human agreement metrics and qualitatively investigate how and why LLMs fail. We also propose actionable recommendations. We analyzed disagreements between LLMs and researchers across six software engineering SRs and over 1,000 primary study papers. For each SR, papers were screened independently by human experts and LLMs in zero-shot mode, resulting in Kappa values ranging from 0.52 to 0.77. Qualitative analysis suggests that human-LLM disagreement results from recurring, identifiable causes, such as boundary ambiguity in key terms, keyword overemphasization, and incorrect topic inference. Based on these findings, we propose recommendations such as validating semantic understandi","title":"Understanding LLMs in Title-Abstract Screening: From Disagreements to Recommendations","url":"https://arxiv.org/abs/2606.17588","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.17588v1 Announce Type: cross \nAbstract: Several studies have examined the use of large language models (LLMs) for title-abstract screening in systematic reviews (SRs), reporting mixed accuracy. However, questions of reliability remain largely unaddressed. In this study, we go beyond quantitative LLM-human agreement metrics and qualitatively investigate how and why LLMs fail. We also propose actionable recommendations. We analyzed disagreements between LLMs and researchers across six software engineering SRs and over 1,000 primary study papers. For each SR, papers were screened independently by human experts and LLMs in zero-shot mode, resulting in Kappa values ranging from 0.52 to 0.77. Qualitative analysis suggests that human-LLM disagreement results from recurring, identifiable causes, such as boundary ambiguity in key terms, keyword overemphasization, and incorrect topic inference. Based on these findings, we propose recommendations such as validating semantic understandi","title":"Understanding LLMs in Title-Abstract Screening: From Disagreements to Recommendations","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-17T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.17588"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:293e3ef3d2c64b8f6eacdfbd6be35668aa25b1c042d5911e14c5a9a637865095ea0563f753a0a2e682f6d9fddcf300a0a8ae76e28c80a54e107dacd7c99ff603","signer":"crovia.substrate","subject":{"observed_at":"2026-06-17T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.17588"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"1cca8bfae37807fd52995254af264e339696513fe539ec1a5d5853d6017c67b6","leaf_index":230984,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"d70bb64e37034c808289ebbb300eb66176e5aaf1b1d02d56adceabebea1a1b80","side":"right"},{"sibling":"e92220b7a8db33b04adb8ad14e555244c0e4665ba013cd0ecbe3c17cfebe9d51","side":"right"},{"sibling":"8ccb515329ea1cf97f75b6ba9ba43d4d2d1ce6e8e606523f2d8456a1c8276a93","side":"right"},{"sibling":"75ee4510e350ca59c448a67b4abfca1071f5a5e6936cee512b00d97355a3c9a3","side":"left"},{"sibling":"285cfbd8ed72a94107849452b5ce5c8e1b0141747ba7e227004fca92993e516e","side":"right"},{"sibling":"b81e03292fa07ff830b444c07ef37ec24747d2035531f28f436dbc6110f3590d","side":"right"},{"sibling":"8daceae0255c786ffdef12a608da0c1a09b9a2ceedfcc067120bfd86b4ab9952","side":"left"},{"sibling":"ce828166a6a4fc2ff2681c985a56053a1f8b683cc245e0b232fa5939310b3ad9","side":"right"},{"sibling":"ebbec9ce4bf43a3f5f71e2c07df4c91301abefa2da727a4d748099a74e980bc3","side":"right"},{"sibling":"1d74941c32baeab8cad08f8700cb49427d8256f231534f2a225b2bb3e84e4ff8","side":"left"},{"sibling":"d5b9f8b1a2c9f6a46e17982dfbe6ce1f3b5fa4e730220397f2253d114dcc8486","side":"left"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_a7717822c7d16502e9785d3eeb267386bb7aa3d5803a1f09b2d8d100b1e6d968"}}