{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_1f68795c337a7674ae9f564c0d1c47c458a5a3ad77592d3be5f6efe87ca95ec6","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_1f68795c337a7674ae9f564c0d1c47c458a5a3ad77592d3be5f6efe87ca95ec6","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6fdf0f9d56901ce15b29d59f94aa8e450de02558ae6ab481342851188d99120b","published":"Wed, 17 Jun 2026 00:00:00 -0400","receipt_hash":"6fdf0f9d56901ce15b29d59f94aa8e450de02558ae6ab481342851188d99120b","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6fdf0f9d56901ce15b29d59f94aa8e450de02558ae6ab481342851188d99120b","observed_at":"2026-06-17T04:43:17.968423Z","parent_run_hash":"8f56c4deb22b28178ba7974d6ffc5ff17336d44c95dd94de80095705245fa113","published":"Wed, 17 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.17165v1 Announce Type: cross \nAbstract: Organizations and researchers show increasing interest in using large language models (LLMs) in place of human participants in A/B tests, in the hope of experimenting faster and at lower cost. We study when a treatment effect estimated on LLM outcomes recovers the effect that would have been measured on the human population of interest. Distributional equivalence between LLM and human outcomes would make any standard estimator valid but is unrealistic. We therefore develop a statistical framework that adapts surrogate endpoint theory to LLMs. The framework shows that calibrating LLM outcomes to human outcomes identifies the average treatment effect under surrogacy and comparability conditions that are jointly weaker than distributional equivalence. When these conditions fail, the effect of interest is only partially identified, and we provide diagnostics that can falsify surrogacy on historical experiments together with a bound on the ","title":"Statistical Foundations of LLM-based A/B Testing: A Surrogacy Framework for Human Causal Inference","url":"https://arxiv.org/abs/2606.17165","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.17165v1 Announce Type: cross \nAbstract: Organizations and researchers show increasing interest in using large language models (LLMs) in place of human participants in A/B tests, in the hope of experimenting faster and at lower cost. We study when a treatment effect estimated on LLM outcomes recovers the effect that would have been measured on the human population of interest. Distributional equivalence between LLM and human outcomes would make any standard estimator valid but is unrealistic. We therefore develop a statistical framework that adapts surrogate endpoint theory to LLMs. The framework shows that calibrating LLM outcomes to human outcomes identifies the average treatment effect under surrogacy and comparability conditions that are jointly weaker than distributional equivalence. When these conditions fail, the effect of interest is only partially identified, and we provide diagnostics that can falsify surrogacy on historical experiments together with a bound on the ","title":"Statistical Foundations of LLM-based A/B Testing: A Surrogacy Framework for Human Causal Inference","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-17T04:43:17Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.17165"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:ee12a111f07958f4f86b6284cea294a1d1f458597fe6e3af40900ae043feaf002929126aae8b1266b549c45df53129ad20056bc6f39f2435e79be9bf4eaab009","signer":"crovia.substrate","subject":{"observed_at":"2026-06-17T04:43:17Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.17165"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"d9228fb8206a572ff248bf773ae360dbd7a3cd8609d1d6c2fcc5c45eca30a475","leaf_index":230934,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"ec1121ed80eeb43c5f1ae84ff6df5b8e6b8471302a996cd4a483df2dbcbb8b83","side":"right"},{"sibling":"95c8c6c60c41b2555839e4f1f5e6614f61b18b75d46e48b359a28806969229c6","side":"left"},{"sibling":"e11cdf369942a472333ce9da1b37db62cb0bec019c024a86f023fb56d4b6c053","side":"left"},{"sibling":"fe1c34d8743f0a6edd886782a55e9b5340a3ae68af354b5b98cee2b8b4a7c911","side":"right"},{"sibling":"89058150059fa791f48320ccd96dcf113012f88382777ba50eac9c86fc142d70","side":"left"},{"sibling":"53a58806ad794dc7abdba5c5adeb9a0ba8e12952d6a841d4180adb65f356cedf","side":"right"},{"sibling":"46a8ccd2dac1ed36480a55cc7afe558ddcd720bdcdac4bb787046b5795ed8399","side":"right"},{"sibling":"ce828166a6a4fc2ff2681c985a56053a1f8b683cc245e0b232fa5939310b3ad9","side":"right"},{"sibling":"ebbec9ce4bf43a3f5f71e2c07df4c91301abefa2da727a4d748099a74e980bc3","side":"right"},{"sibling":"1d74941c32baeab8cad08f8700cb49427d8256f231534f2a225b2bb3e84e4ff8","side":"left"},{"sibling":"d5b9f8b1a2c9f6a46e17982dfbe6ce1f3b5fa4e730220397f2253d114dcc8486","side":"left"},{"sibling":"d10d772a4984cae00e65ab24af21d1d260e475bbaae3f17871859eb705bd3999","side":"right"},{"sibling":"e79159853f2f35ddae8e3247d515e433c534277b287d65bbd77ae989aa4992fa","side":"right"},{"sibling":"3054319f1840cce0eaaf0bc4b1ae38e5bf8b6210927d924a750775cc7d77cca6","side":"right"},{"sibling":"0fd8b5059f279c4a4a6688de2472fdbc25543fee183df19b21dacca880354cff","side":"right"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":232015,"merkle_root":"62bfb7809bb55667ad7eeebdb48267b9b2c1ee89bb808ea1e6d3a814a15aa402","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260617T133701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-17T13:39:59Z","sig_algorithm":"ed25519","signature":"930a3563c643cc7518d048b12a1f5392a96a5edce49533ba0a56f11b6cc69319bb1c800aacc46b52a6941c4d05dae255a45cac757d6093c971b96852c24f140e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_1f68795c337a7674ae9f564c0d1c47c458a5a3ad77592d3be5f6efe87ca95ec6"}}