{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_5afec8d158f647227e554f2a0b7f3ea23a330c1ea6e53f4390a699a1dce14288","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_5afec8d158f647227e554f2a0b7f3ea23a330c1ea6e53f4390a699a1dce14288","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"c3f26b05dc5cd04bc9761ced6e80e8cb33b278454adb67e37571d38c1f476044","published":"Tue, 02 Jun 2026 00:00:00 -0400","receipt_hash":"c3f26b05dc5cd04bc9761ced6e80e8cb33b278454adb67e37571d38c1f476044","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"c3f26b05dc5cd04bc9761ced6e80e8cb33b278454adb67e37571d38c1f476044","observed_at":"2026-06-02T04:43:38.825628Z","parent_run_hash":"c2a9665c814770d56765bb764e6a6c7e4fa7d4e9708e157ca0f7440c89927d54","published":"Tue, 02 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2603.23485v2 Announce Type: replace-cross \nAbstract: Standard evaluation practices assume that large language model (LLM) outputs are stable when prompts are embedded in contextually equivalent discourses. Here, we test this assumption in the setting of gender inference. Using a controlled pronoun selection task, we introduce minimal, theoretically uninformative discourse context and find that this induces large, systematic shifts in model outputs. Correlations with cultural gender stereotypes, present in decontextualized settings, weaken or disappear once context is introduced, while theoretically irrelevant features, such as the gender of a pronoun for an unrelated referent, become the most informative predictors of model behavior. A Contextuality-by-Default analysis reveals that, in 19--52\\% of cases across models, this dependence persists after accounting for all marginal effects of context on individual outputs and cannot be attributed to simple pronoun repetition. These fin","title":"Failure of contextual invariance in large language models","url":"https://arxiv.org/abs/2603.23485","vendor":"arxiv_cs_ai"},"summary":"arXiv:2603.23485v2 Announce Type: replace-cross \nAbstract: Standard evaluation practices assume that large language model (LLM) outputs are stable when prompts are embedded in contextually equivalent discourses. Here, we test this assumption in the setting of gender inference. Using a controlled pronoun selection task, we introduce minimal, theoretically uninformative discourse context and find that this induces large, systematic shifts in model outputs. Correlations with cultural gender stereotypes, present in decontextualized settings, weaken or disappear once context is introduced, while theoretically irrelevant features, such as the gender of a pronoun for an unrelated referent, become the most informative predictors of model behavior. A Contextuality-by-Default analysis reveals that, in 19--52\\% of cases across models, this dependence persists after accounting for all marginal effects of context on individual outputs and cannot be attributed to simple pronoun repetition. These fin","title":"Failure of contextual invariance in large language models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-02T04:43:38Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2603.23485"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:f268f92d2cb0a506f1e9c56c806d2c4748f647722a7bfb7dc571a180006433265fca9154061cb35940e7fc01e5108f105ae0bf716c1e6beb8d1b40f8e1e3ce0d","signer":"crovia.substrate","subject":{"observed_at":"2026-06-02T04:43:38Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2603.23485"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"fb6924dc29e9a480a5273295fc8914ee3d5f08d7f36d23efcfd5f55df63cc886","leaf_index":206014,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"86afeabbf2ee0c43490a7ce667f7171e39c1e9d3a17bd6bffdecbd410d4d3775","side":"right"},{"sibling":"b7f54b3ebdab3b6e3862ee02133492ac4d645a6594f12dcedb09c1d8403d4f49","side":"left"},{"sibling":"fdbfd3341d80e9d18897a552d43260544004b45d7d28335274108131c90d0a68","side":"left"},{"sibling":"6b0be801f47477655e178bbcdf4ae4d76c7410b65bbbd456ac03c93d0ebb058d","side":"left"},{"sibling":"ef7822a139a81661feb4ef03136b22cac59ad14f2042744dc76099886b7da904","side":"left"},{"sibling":"a003ca73509f4616e19ed77ec7fc4ca125a94b735b3d601c3d146997158b1af5","side":"left"},{"sibling":"9d3cdcb044db8a2b1007e70fe19fdb164ed8f48d6165a457b3e5a30bd96cb82f","side":"right"},{"sibling":"a87881f2c8c65f06a7d79e20af3b2022c798d3fe9dfa79b997557c49e1803708","side":"left"},{"sibling":"6ac554ede1f78dc3a28ebe5e1155c2b9c258b71805fdd7ad2ec992697f3ccd0c","side":"right"},{"sibling":"5e1949edfad76bc6a73d008dc8be8a0c6fe4a7fb64c8beaf70e5023ca11d35e0","side":"right"},{"sibling":"e4bd1aaaf3f336d9b072b4d1fc234246fc3cf2874ca873b7741c03eaf97911f2","side":"left"},{"sibling":"e6adead8216db4cae92f0a036d53baebf30eed95a99c0d10758aa75bb7780f2f","side":"right"},{"sibling":"1acc2b7ff453ffd8c97b80ae4db5358780f0c6796874fd75403791dbe99f8cd7","side":"right"},{"sibling":"24d1bb4b13e0e46131b27b70a48e65fcf4e2e14b95e3bb83ade821e9df530f6b","side":"left"},{"sibling":"5f86f58c28b1a86ae06dfff4666bb9fba8866021a81fd4f1d200aa9af4722dfb","side":"right"},{"sibling":"f6cc6f94f6944ae21390afc65ac9e91dc31f84ee6e060681bba5ae08058294bd","side":"right"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":206226,"merkle_root":"d2a6d32b13cbf343fb143b21a756d0533864ae6577a376ee84ba867b949207ec","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260602T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-02T05:37:46Z","sig_algorithm":"ed25519","signature":"abd9956cfb19dd1fb8142c46a220bac2514848c6abb0e79b8b0940206cc3ebb00894d4daaf9f786427f82a7cc12482e7fda79054ebb06bceaa9b4b97e23fb30e","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_5afec8d158f647227e554f2a0b7f3ea23a330c1ea6e53f4390a699a1dce14288"}}