{"_canonicalization":{"envelope_id":"axm_ + sha256(envelope minus {signature, axiom_id, anchors})","envelope_signature":"ed25519(envelope minus {signature, axiom_id})","json":"sort_keys=True, separators=(',',':'), ensure_ascii=False, allow_nan=False, utf-8","leaf_hash":"sha256(0x00 || canonical_json(envelope_full))","seal_signature":"ed25519(seal minus {signature, sig_algorithm})"},"axiom_id":"axm_b391f708772bac652d9c247a8a0afeaccc2c9cb883591c6c0d180747f0730fba","bitcoin_anchor":{"bitcoin_attestations":[],"calendar_attestations":[],"ots_url":"","stamped_at":"","status":"pending_next_stamp"},"envelope":{"anchors":[{"chain":"crovia.axiom_graph","height":0,"merkle_proof":"spider_vendor_press_v1","root_at_anchor":"spider_vendor_press_v1"}],"axiom_id":"axm_b391f708772bac652d9c247a8a0afeaccc2c9cb883591c6c0d180747f0730fba","axiom_type":"AX.OBS","body":{"axiom_subtype":"news.vendor_press.v1","category":"news","fingerprint":"329f0b625d090b33eb1bec88048d7cbe501a5eb00ef57cdf713ef41506625f6a","published":"Fri, 10 Jul 2026 00:00:00 -0400","receipt_hash":"329f0b625d090b33eb1bec88048d7cbe501a5eb00ef57cdf713ef41506625f6a","schema":"spider.news.vendor_press.v1","spider":"vendor_press","spider_record":{"axiom_subtype":"news.vendor_press.v1","category":"news","decision_hint":"POSITIVE","envelope_target":"AX.OBS","fingerprint":"329f0b625d090b33eb1bec88048d7cbe501a5eb00ef57cdf713ef41506625f6a","observed_at":"2026-07-10T04:43:53.465232Z","parent_run_hash":"06997be187ba20932a2030c56de194579eacf484085252a25bee544eab183e91","published":"Fri, 10 Jul 2026 00:00:00 -0400","runtime_version":"0.1.0","schema":"spider.news.vendor_press.v1","source_status":200,"source_url":"https://export.arxiv.org/rss/cs.AI","spider":"vendor_press","summary_excerpt":"arXiv:2607.07903v1 Announce Type: cross \nAbstract: Large language models (LLMs) exhibit remarkable capabilities but remain highly vulnerable to adversarial prompts and jailbreak attacks. Existing approaches primarily analyze these failures through input-output behaviors or attribution methods, offering limited insight into how adversarial perturbations alter the model's internal reasoning. Consequently, the mechanisms underlying unsafe or incorrect behaviors remain poorly understood. We introduce a mechanistic framework for diagnosing LLM vulnerabilities using paired internal computation graphs, which represent prompt-specific inference as structured causal interactions among latent features. By constructing and aligning computation graphs for clean and attacked prompts, we reveal that adversarial attacks induce systematic transformations of internal reasoning, including suppression of safety-relevant components, emergence of attack-specific features, and rerouting of computation paths","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","url":"https://arxiv.org/abs/2607.07903","vendor":"arxiv_cs_ai"},"summary":"arXiv:2607.07903v1 Announce Type: cross \nAbstract: Large language models (LLMs) exhibit remarkable capabilities but remain highly vulnerable to adversarial prompts and jailbreak attacks. Existing approaches primarily analyze these failures through input-output behaviors or attribution methods, offering limited insight into how adversarial perturbations alter the model's internal reasoning. Consequently, the mechanisms underlying unsafe or incorrect behaviors remain poorly understood. We introduce a mechanistic framework for diagnosing LLM vulnerabilities using paired internal computation graphs, which represent prompt-specific inference as structured causal interactions among latent features. By constructing and aligning computation graphs for clean and attacked prompts, we reveal that adversarial attacks induce systematic transformations of internal reasoning, including suppression of safety-relevant components, emergence of attack-specific features, and rerouting of computation paths","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","vendor":"arxiv_cs_ai"},"confidence":{"method":"deterministic"},"decision":"POSITIVE","issued_at":"2026-07-10T04:43:53Z","notes":"Spider vendor_press (news) news.vendor_press.v1","object":{"captured_by":"crovia.spider.vendor_press","primary_source_url":"https://arxiv.org/abs/2607.07903"},"predecessors":[],"schema":"crovia.axiom.v1","signature":"ed25519:8ee3cc7c3f3bcef217b06ce0ad1383c46d91c78789347b7ffca1c6c747dc5f3291377180a5d29b9e90b453e50129120ce5c90d2ab64c0ba96e257a7bf8f72001","signer":"crovia.substrate","subject":{"observed_at":"2026-07-10T04:43:53Z","source_collector":"spider:vendor_press","target_id":"https://arxiv.org/abs/2607.07903"},"tsa":{"authority":"crovia.substrate.bootstrap","rfc3161_token":"{\"kind\":\"crovia.bootstrap.tsa\",\"source_jsonl\":\"/opt/crovia/spider/data/news/vendor_press_v1.jsonl\",\"source_seal_merkle_root\":\"spider_vendor_press_v1\",\"upgrade_path\":\"Sessione H \\u2014 OpenTimestamps weekly anchor\"}"},"zk_mode":"clear","zk_proof":null},"ledger":{"leaf_hash":"efaec127f49f11987cbebc8f77067b3e7280d7c51fc074b82f8cad792d785387","leaf_index":299428,"ledger_path":"/opt/crovia/substrate/axiom_ledger.jsonl"},"merkle_proof":{"hash_alg":"sha256","leaf_prefix":"0x00","node_prefix":"0x01","odd_leaf_rule":"duplicate_last","path":[{"sibling":"3cc5d45031670525106a70e8fecab5b99493d9ceab08892a1c25fd8122807f7f","side":"right"},{"sibling":"a76dcfad4217b88cfdc8a2de0de55749464f0c9d7ad1072975a3d26dddf833fa","side":"right"},{"sibling":"a643c59eb6c8c29006b1615fe34e9c0a665573e2b7efa875a3938329a1f48d40","side":"left"},{"sibling":"64350c7120cd7bace59c739de8174fe7446b61038d57373271d3e82633e0db70","side":"right"},{"sibling":"ad3a5659d3356b9ed92da123d037d5a3f0bcab19684f22c114fc2337695a08a2","side":"right"},{"sibling":"f28ac6bb8a391ac9d62ae57f3339b85726df8e9410d52e586ea8c062e2e7cdb1","side":"left"},{"sibling":"315f14e236d484d058208a092c7e486bc56d42850a756f3d89959981a1458ce1","side":"right"},{"sibling":"b1b72243c904808d4df9b283c65832a729a411520c97fd9d17f891f214971a0a","side":"left"},{"sibling":"2c5ce75ad134aa442d7fbb5dc9968b0f2e48097c2ffaa2111d405de1c4a9c48b","side":"left"},{"sibling":"36a2c507282befad57202dd10278a66d37b402692f195a22d2275b3d1b2488d8","side":"right"},{"sibling":"f80e8d47e0860527b906fc2dba9a52609f7f623f2772479ae922cc019bab36d9","side":"right"},{"sibling":"cae83500ab2c25555aa6b5eaf9232696d15a868d91b34f7531dd955daadf70f7","side":"right"},{"sibling":"64dab64d51bdcb909e2a5e37efb8909d6704ecf824be484b5d2b60ee6e518890","side":"left"},{"sibling":"576f134a23c19a758ae5efd53016092a74b9900e878cf6eb4f3dab6be682b395","side":"right"},{"sibling":"3c65f53d7c3e4feba7c745e8df1327760ffa768eec84336db14d515a31731532","side":"right"},{"sibling":"d8b9143917b539c543cf4448cec00131f8b807bd8004979c54ebe09798748c66","side":"left"},{"sibling":"8728642cdb98496d916cc653f0d919d7eb2e89d0c529927c9e89091074ad584c","side":"right"},{"sibling":"8025674cb002a22ae243ca0c295c18c1d0ee119189ea88e08ac14a3a1468b8e3","side":"right"},{"sibling":"1cecb7f447febd025aac272837c80de218aecc6485d2395a509b2a1f1b9c746e","side":"left"}]},"schema":"crovia.axiom_proof.v1","seal":{"first_collector_run_id":"","first_receipt_hash":"","jsonl_path":"/opt/crovia/substrate/axiom_ledger.jsonl","key_id":"430895f101d38164","last_collector_run_id":"","last_receipt_hash":"","leaf_count":299721,"merkle_root":"f7115d63193d3285ca28cb9f741ecea2513f9b3e492e785f97076f3cf8f9bb98","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","run_id":"hourly_json_retrofit_20260710T053701Z","schema":"crovia.seal.v1","seal_family_version":"crovia-seal-family/1","seal_kind":"substrate_batch","sealed_at":"2026-07-10T05:38:19Z","sig_algorithm":"ed25519","signature":"4eeedeb744885bff6523d66b1cb86bde62b36917696367addfd980d90b31011daece2e01b20608e4c0870ec31dd0bcb57d297447f01f153bf0716ad527142b00","signer_version":"1.1.0"},"trust_root":{"key_id":"430895f101d38164","public_key_hex":"cf742e26f75669dc673cb5c0786a1ae23ae8ca19c347317192ce40c28a7ff25c","signature_algorithm":"ed25519","url":"/registry/canon/TRUST_ROOT.md"},"verifier":{"spec":"/registry/canon/AXIOM_RECEIPT_v1.md","url":"/v/axm_b391f708772bac652d9c247a8a0afeaccc2c9cb883591c6c0d180747f0730fba"}}