{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b44d91c014f7fcbf7819c031f19c4fbd6f6431bb36ff8efbb3827084f857e02f","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b44d91c014f7fcbf7819c031f19c4fbd6f6431bb36ff8efbb3827084f857e02f","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"842b761ca7c5bb622c193ae76fd12e43b6347e50b90a1f42b46057351d0a90ed","published":"Fri, 26 Jun 2026 00:00:00 -0400","receipt_hash":"842b761ca7c5bb622c193ae76fd12e43b6347e50b90a1f42b46057351d0a90ed","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"842b761ca7c5bb622c193ae76fd12e43b6347e50b90a1f42b46057351d0a90ed","observed_at":"2026-06-26T04:43:58.168958Z","parent_run_hash":"9459505a803125e4b968df08c74ed0054a2aafd44e4e1a036e3b0709a8a65cb4","published":"Fri, 26 Jun 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2606.09843v2 Announce Type: replace-cross \nAbstract: Large language models (LLMs) give stable answers to personality questionnaires, yet these self-reports fail to predict how the models actually behave. Is this gap an artifact of forcing human trait categories onto LLMs, or something deeper about LLM self-report itself? To find out, we built the first psychometric instrument whose dimensions are derived bottom-up from LLM behavior rather than borrowed from human psychology. Administering 300 items (240 Likert + 60 scenario) to 25 LLMs across 17 model families, 30 times each, exploratory factor analysis revealed five replicable, highly reliable factors: Responsiveness, Deference, Boldness, Guardedness, and Verbosity (all Tucker $\\phi \\geq .957$, all $\\alpha \\geq .930$). We then collected 2,500 open-ended behavioral samples and had them rated by 151 humans and a three-judge LLM ensemble. Humans and judges agreed about model behavior ($\\bar{r} = .51$), but self-report predicted nei","title":"An LLM-Native Psychometric Instrument Does Not Predict LLM Behavior: Evidence Across 25 Models","url":"https://arxiv.org/abs/2606.09843","vendor":"arxiv_cs_ai"},"summary":"arXiv:2606.09843v2 Announce Type: replace-cross \nAbstract: Large language models (LLMs) give stable answers to personality questionnaires, yet these self-reports fail to predict how the models actually behave. Is this gap an artifact of forcing human trait categories onto LLMs, or something deeper about LLM self-report itself? To find out, we built the first psychometric instrument whose dimensions are derived bottom-up from LLM behavior rather than borrowed from human psychology. Administering 300 items (240 Likert + 60 scenario) to 25 LLMs across 17 model families, 30 times each, exploratory factor analysis revealed five replicable, highly reliable factors: Responsiveness, Deference, Boldness, Guardedness, and Verbosity (all Tucker $\\phi \\geq .957$, all $\\alpha \\geq .930$). We then collected 2,500 open-ended behavioral samples and had them rated by 151 humans and a three-judge LLM ensemble. Humans and judges agreed about model behavior ($\\bar{r} = .51$), but self-report predicted nei","title":"An LLM-Native Psychometric Instrument Does Not Predict LLM Behavior: Evidence Across 25 Models","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-06-26T04:43:58Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2606.09843"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:db7810528e107addd735e6dd8e5542d936cb15d3c59fdca3784f65674fb0eaf19494586fbef76a97136454bc08f3d91abc5be57fbbd9c723d08dd7a7154e880e","signer":"crovia.substrate","subject":{"observed_at":"2026-06-26T04:43:58Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2606.09843"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"14a8545eb4741e48bef1288bffcef4ae19940e6b9a471067b8fd9cb442fc7163","leaf_index":251252,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"ceae3b5ffe2bf2e445b6b515f384a49448b057cb69b179d57474c676dfbd2489","side":"right"},{"sibling":"b3466d6ba1fdd37c94f5cb0b5a7e09eeeb596103056ef6b7c6d84ec0cf21310c","side":"right"},{"sibling":"648af873b7ff88652c0ef9723657f1c17d31004aa5c864f94c0aef366ae953b4","side":"left"},{"sibling":"62868a6170911feb7c62f02e9053b4bd73e45c74482afa7ee8b1877823706e3f","side":"right"},{"sibling":"973238270d3540b2e0c3af63d14225829dc18bfc2ceedb63cb72faeedcdabde7","side":"left"},{"sibling":"51263142806731dfcb141f012b12b625e80221cded7cd2fe31f3f94c663cb98d","side":"left"},{"sibling":"7de9e0cbfd7739efcc9d34b6ad0610c6607927b8ed4fcecf3bc616e114ce53c4","side":"left"},{"sibling":"846fe23f5401fbe8ae1dd175eade4cb955db0a1adde862418a9a781a48f6283c","side":"right"},{"sibling":"d097e1d1d25acb37e74792d46ad69ba853039e2f2b33a092c9ea32a837aeed84","side":"left"},{"sibling":"b6709caadc8510310ee2ec0b66d1058fcad31c91c65cc2bad6f46a693d553580","side":"right"},{"sibling":"a72c3b8804a37d1a9d18e02e6fdb048bc8ea6b0746909cd2de10bdbabc793737","side":"left"},{"sibling":"803703dc2c50a646fa77b57c0056e9a5126611ba4bcde0d6013ccd6b2d44bdbf","side":"right"},{"sibling":"e78f244b1b8df6d5e3fdc6dd76b5c27d4e6fe3b93b8cd61355497b63d7e4cfe8","side":"left"},{"sibling":"6167cb552ed6871fbf0afcf3db01d1017af7d472b136fcbe5404d1df09f41cc1","side":"right"},{"sibling":"f29798d8bb6aa9900eab878992d9ff0c53266debd87472f31ab26a6a3fb55880","side":"left"},{"sibling":"a04392fb9f2a3a620840e3b3fecd93d12e6d224e481c162389d8ae64e7194599","side":"left"},{"sibling":"c300cf0154c136afc09b1702a0be98f4ba5b6dc5cf57e8cc714ec1eaf4196eff","side":"left"},{"sibling":"d841ad93efda0869e5eb97678f348f03f5caab4353e05ff4bf18f47fb945b822","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":251380,"merkle_root":"e042805d07cd8dc777d49695ad78b8d4ec9721df271ff0245d7706773c30b4a5","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260626T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-06-26T05:37:59Z","sig_algorithm":"ed25519","signature":"c68a6e827804771acd244208495c5e35c6f417307db3c76192f038dedaa5b5f86e019074a3bea353357ff7457924c4f7907832638bb9a4d3d18e7a7f50c6a10b","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b44d91c014f7fcbf7819c031f19c4fbd6f6431bb36ff8efbb3827084f857e02f"}}