{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_373053a0159c4a2419a1cbcca8c764daf35996c70b832b42a03fd99b73687665","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_373053a0159c4a2419a1cbcca8c764daf35996c70b832b42a03fd99b73687665","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"6e7ad95b97c83ef47fb52dca71f322aac872eb803ecf49ec6ea6409ad02a0781","published":"Fri, 10 Jul 2026 00:00:00 -0400","receipt_hash":"6e7ad95b97c83ef47fb52dca71f322aac872eb803ecf49ec6ea6409ad02a0781","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"6e7ad95b97c83ef47fb52dca71f322aac872eb803ecf49ec6ea6409ad02a0781","observed_at":"2026-07-10T04:43:53.465232Z","parent_run_hash":"06997be187ba20932a2030c56de194579eacf484085252a25bee544eab183e91","published":"Fri, 10 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.07918v1 Announce Type: cross \nAbstract: Current safety methods for large language models are known to be vulnerable to adversarial attacks, motivating research into robust alternatives. Latent Adversarial Training (LAT) is among the most effective defenses, but can degrade utility and requires training on large datasets of harmful prompts. We introduce Latent Personality Alignment (LPA), which replaces explicit harm refusal with adversarial training on just 66 harm-agnostic statements drawn from psychometric personality literature. We hypothesize that personality-anchored representations share latent structure with harm avoidance, so adversarially stabilizing them implicitly constrains the subspace exploited by jailbreak attacks. LPA achieves near-zero attack success rates on HarmBench across direct requests and five jailbreak methods, despite never seeing harmful content during training and no loss of performance on standard benchmarks. Moreover, the training process is lig","title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","url":"https://arxiv.org/abs/2607.07918","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.07918v1 Announce Type: cross \nAbstract: Current safety methods for large language models are known to be vulnerable to adversarial attacks, motivating research into robust alternatives. Latent Adversarial Training (LAT) is among the most effective defenses, but can degrade utility and requires training on large datasets of harmful prompts. We introduce Latent Personality Alignment (LPA), which replaces explicit harm refusal with adversarial training on just 66 harm-agnostic statements drawn from psychometric personality literature. We hypothesize that personality-anchored representations share latent structure with harm avoidance, so adversarially stabilizing them implicitly constrains the subspace exploited by jailbreak attacks. LPA achieves near-zero attack success rates on HarmBench across direct requests and five jailbreak methods, despite never seeing harmful content during training and no loss of performance on standard benchmarks. Moreover, the training process is lig","title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-10T04:43:53Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.07918"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:a8503a407aa7ffbe9eb9c2f4a442464d2458c74295a9f9dffe59909070aa3516c8b4d2f827e1c62a68a555d36d0d8c1472dba147d75de67bb16f0e1688c5950a","signer":"crovia.substrate","subject":{"observed_at":"2026-07-10T04:43:53Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.07918"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"6d9d590a15d92caa16c40056cf9c78c7268abf8a7027e9dae864469718171242","leaf_index":299430,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"c5b756fbe7b1ba8653a09ee6cbb7e0d88b9069b6bd69f2b87144e29c6d8f4536","side":"right"},{"sibling":"40368199ac1f8a960660d1edb3e671f952aba7a72c69561c4e8d88d40116bc37","side":"left"},{"sibling":"a643c59eb6c8c29006b1615fe34e9c0a665573e2b7efa875a3938329a1f48d40","side":"left"},{"sibling":"64350c7120cd7bace59c739de8174fe7446b61038d57373271d3e82633e0db70","side":"right"},{"sibling":"ad3a5659d3356b9ed92da123d037d5a3f0bcab19684f22c114fc2337695a08a2","side":"right"},{"sibling":"f28ac6bb8a391ac9d62ae57f3339b85726df8e9410d52e586ea8c062e2e7cdb1","side":"left"},{"sibling":"315f14e236d484d058208a092c7e486bc56d42850a756f3d89959981a1458ce1","side":"right"},{"sibling":"b1b72243c904808d4df9b283c65832a729a411520c97fd9d17f891f214971a0a","side":"left"},{"sibling":"2c5ce75ad134aa442d7fbb5dc9968b0f2e48097c2ffaa2111d405de1c4a9c48b","side":"left"},{"sibling":"36a2c507282befad57202dd10278a66d37b402692f195a22d2275b3d1b2488d8","side":"right"},{"sibling":"f80e8d47e0860527b906fc2dba9a52609f7f623f2772479ae922cc019bab36d9","side":"right"},{"sibling":"cae83500ab2c25555aa6b5eaf9232696d15a868d91b34f7531dd955daadf70f7","side":"right"},{"sibling":"64dab64d51bdcb909e2a5e37efb8909d6704ecf824be484b5d2b60ee6e518890","side":"left"},{"sibling":"576f134a23c19a758ae5efd53016092a74b9900e878cf6eb4f3dab6be682b395","side":"right"},{"sibling":"3c65f53d7c3e4feba7c745e8df1327760ffa768eec84336db14d515a31731532","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"8728642cdb98496d916cc653f0d919d7eb2e89d0c529927c9e89091074ad584c","side":"right"},{"sibling":"8025674cb002a22ae243ca0c295c18c1d0ee119189ea88e08ac14a3a1468b8e3","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":299721,"merkle_root":"f7115d63193d3285ca28cb9f741ecea2513f9b3e492e785f97076f3cf8f9bb98","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260710T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-10T05:38:19Z","sig_algorithm":"ed25519","signature":"4eeedeb744885bff6523d66b1cb86bde62b36917696367addfd980d90b31011daece2e01b20608e4c0870ec31dd0bcb57d297447f01f153bf0716ad527142b00","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_373053a0159c4a2419a1cbcca8c764daf35996c70b832b42a03fd99b73687665"}}