{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M2YG2BONZ4YXVWSJS32IDTXT5O","short_pith_number":"pith:M2YG2BON","schema_version":"1.0","canonical_sha256":"66b06d05cdcf317ada4996f481cef3ebaaaff6bc785ab32db9279a6af82d6439","source":{"kind":"arxiv","id":"2506.11618","version":2},"attestation_state":"computed","paper":{"title":"Convergent Linear Representations of Emergent Misalignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anna Soligo, Edward Turner, Neel Nanda, Senthooran Rajamanoharan","submitted_at":"2025-06-13T09:39:54Z","abstract_excerpt":"Fine-tuning large language models on narrow datasets can cause them to develop broadly misaligned behaviours: a phenomena known as emergent misalignment. However, the mechanisms underlying this misalignment, and why it generalizes beyond the training domain, are poorly understood, demonstrating critical gaps in our knowledge of model alignment. In this work, we train and study a minimal model organism which uses just 9 rank-1 adapters to emergently misalign Qwen2.5-14B-Instruct. Studying this, we find that different emergently misaligned models converge to similar representations of misalignme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11618","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-13T09:39:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cc860aeae99a20b07163f62ce61c6d9b7ad54277edbea8d3b44f18214531feb8","abstract_canon_sha256":"a12f503049a943cedcb626d29d9779ad1c7d54392423308750220cd10bb22559"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:53.517625Z","signature_b64":"c/MEyZBulP+33KvzfdllrwVV1xQBHCgm4QRwwOk1aMryDnr6vCrMTeZ6c580o4AqreF5qz4+WhNYmpJuK9cqAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66b06d05cdcf317ada4996f481cef3ebaaaff6bc785ab32db9279a6af82d6439","last_reissued_at":"2026-07-05T11:24:53.517116Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:53.517116Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Convergent Linear Representations of Emergent Misalignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anna Soligo, Edward Turner, Neel Nanda, Senthooran Rajamanoharan","submitted_at":"2025-06-13T09:39:54Z","abstract_excerpt":"Fine-tuning large language models on narrow datasets can cause them to develop broadly misaligned behaviours: a phenomena known as emergent misalignment. However, the mechanisms underlying this misalignment, and why it generalizes beyond the training domain, are poorly understood, demonstrating critical gaps in our knowledge of model alignment. In this work, we train and study a minimal model organism which uses just 9 rank-1 adapters to emergently misalign Qwen2.5-14B-Instruct. Studying this, we find that different emergently misaligned models converge to similar representations of misalignme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11618","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11618/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11618","created_at":"2026-07-05T11:24:53.517177+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11618v2","created_at":"2026-07-05T11:24:53.517177+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11618","created_at":"2026-07-05T11:24:53.517177+00:00"},{"alias_kind":"pith_short_12","alias_value":"M2YG2BONZ4YX","created_at":"2026-07-05T11:24:53.517177+00:00"},{"alias_kind":"pith_short_16","alias_value":"M2YG2BONZ4YXVWSJ","created_at":"2026-07-05T11:24:53.517177+00:00"},{"alias_kind":"pith_short_8","alias_value":"M2YG2BON","created_at":"2026-07-05T11:24:53.517177+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20225","citing_title":"Actionable Activation Directions for Detecting and Mitigating Emergent Misalignment Across Language Model Families","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09068","citing_title":"Emergent Misalignment Can Be Induced by Sycophancy and Reversed via Alignment Gating","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08682","citing_title":"Activation Steering Induces Emergent Misalignment: A More Comprehensive Evaluation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06223","citing_title":"From Reward-Hack Activations to Agentic Risk States: Context-Calibrated Mechanistic Monitoring in LLM Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06667","citing_title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03810","citing_title":"Consistency Training Can Entrench Misalignment","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00995","citing_title":"Subliminal Learning Is Steering Vector Distillation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07631","citing_title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27676","citing_title":"Unsupervised Identification and Removal of Spurious Correlations During Fine-Tuning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16325","citing_title":"Phase Transitions in Driven Informational Systems: A Two-Field Perspective on Learning Theory and Non-Equilibrium Chemistry","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12798","citing_title":"Emergent and Subliminal Misalignment Through the Lens of Data-Mediated Transfer","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28082","citing_title":"Characterizing the Consistency of the Emergent Misalignment Persona","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00842","citing_title":"Understanding Emergent Misalignment via Feature Superposition Geometry","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19117","citing_title":"LLMs Know They're Wrong and Agree Anyway: The Shared Sycophancy-Lying Circuit","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O","json":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O.json","graph_json":"https://pith.science/api/pith-number/M2YG2BONZ4YXVWSJS32IDTXT5O/graph.json","events_json":"https://pith.science/api/pith-number/M2YG2BONZ4YXVWSJS32IDTXT5O/events.json","paper":"https://pith.science/paper/M2YG2BON"},"agent_actions":{"view_html":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O","download_json":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O.json","view_paper":"https://pith.science/paper/M2YG2BON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11618&json=true","fetch_graph":"https://pith.science/api/pith-number/M2YG2BONZ4YXVWSJS32IDTXT5O/graph.json","fetch_events":"https://pith.science/api/pith-number/M2YG2BONZ4YXVWSJS32IDTXT5O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O/action/storage_attestation","attest_author":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O/action/author_attestation","sign_citation":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O/action/citation_signature","submit_replication":"https://pith.science/pith/M2YG2BONZ4YXVWSJS32IDTXT5O/action/replication_record"}},"created_at":"2026-07-05T11:24:53.517177+00:00","updated_at":"2026-07-05T11:24:53.517177+00:00"}