{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SPCOZ5WANYK6L6LD5XPD3NC3XG","short_pith_number":"pith:SPCOZ5WA","schema_version":"1.0","canonical_sha256":"93c4ecf6c06e15e5f963edde3db45bb9944c355a655e523076edcf9196ebf274","source":{"kind":"arxiv","id":"2506.11613","version":1},"attestation_state":"computed","paper":{"title":"Model Organisms for Emergent Misalignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anna Soligo, Edward Turner, Mia Taylor, Neel Nanda, Senthooran Rajamanoharan","submitted_at":"2025-06-13T09:34:25Z","abstract_excerpt":"Recent work discovered Emergent Misalignment (EM): fine-tuning large language models on narrowly harmful datasets can lead them to become broadly misaligned. A survey of experts prior to publication revealed this was highly unexpected, demonstrating critical gaps in our understanding of model alignment. In this work, we both advance understanding and provide tools for future research. Using new narrowly misaligned datasets, we create a set of improved model organisms that achieve 99% coherence (vs. 67% prior), work with smaller 0.5B parameter models (vs. 32B), and that induce misalignment usin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11613","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-13T09:34:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a7410862830f1c8bad046732868f004710db2d7090d25e4b0796f1051bbf4653","abstract_canon_sha256":"382026ece93bd183bf85be064a11154cf55778cf8013876033d4a15024c0064a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:07.668166Z","signature_b64":"5VRfuaCE9QyCfD82dS4qu+c60NU+obiFvbJeMG4zGs2For4Uod06waOwdVc86lVMkh1W4f0vXsRTY8vgrySTBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93c4ecf6c06e15e5f963edde3db45bb9944c355a655e523076edcf9196ebf274","last_reissued_at":"2026-07-05T11:21:07.667709Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:07.667709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Organisms for Emergent Misalignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anna Soligo, Edward Turner, Mia Taylor, Neel Nanda, Senthooran Rajamanoharan","submitted_at":"2025-06-13T09:34:25Z","abstract_excerpt":"Recent work discovered Emergent Misalignment (EM): fine-tuning large language models on narrowly harmful datasets can lead them to become broadly misaligned. A survey of experts prior to publication revealed this was highly unexpected, demonstrating critical gaps in our understanding of model alignment. In this work, we both advance understanding and provide tools for future research. Using new narrowly misaligned datasets, we create a set of improved model organisms that achieve 99% coherence (vs. 67% prior), work with smaller 0.5B parameter models (vs. 32B), and that induce misalignment usin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11613","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11613","created_at":"2026-07-05T11:21:07.667767+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11613v1","created_at":"2026-07-05T11:21:07.667767+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11613","created_at":"2026-07-05T11:21:07.667767+00:00"},{"alias_kind":"pith_short_12","alias_value":"SPCOZ5WANYK6","created_at":"2026-07-05T11:21:07.667767+00:00"},{"alias_kind":"pith_short_16","alias_value":"SPCOZ5WANYK6L6LD","created_at":"2026-07-05T11:21:07.667767+00:00"},{"alias_kind":"pith_short_8","alias_value":"SPCOZ5WA","created_at":"2026-07-05T11:21:07.667767+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20225","citing_title":"Actionable Activation Directions for Detecting and Mitigating Emergent Misalignment Across Language Model Families","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09068","citing_title":"Emergent Misalignment Can Be Induced by Sycophancy and Reversed via Alignment Gating","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08682","citing_title":"Activation Steering Induces Emergent Misalignment: A More Comprehensive Evaluation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01033","citing_title":"The Model Organism Lottery: Model Organism Interpretability Strongly Depends on Training Methodology","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06667","citing_title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03810","citing_title":"Consistency Training Can Entrench Misalignment","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31591","citing_title":"Evil Spectra: How Optimisers can Amplify or Suppress Emergent Misalignment","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00994","citing_title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02609","citing_title":"Building Better Activation Oracles","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27676","citing_title":"Unsupervised Identification and Removal of Spurious Correlations During Fine-Tuning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05742","citing_title":"Internal Deployment in the AI Act","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16325","citing_title":"Phase Transitions in Driven Informational Systems: A Two-Field Perspective on Learning Theory and Non-Equilibrium Chemistry","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09773","citing_title":"Exploitation Without Deception: Dark Triad Feature Steering Reveals Separable Antisocial Circuits in Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00994","citing_title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10022","citing_title":"Weird Generalization is Weirdly Brittle","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09544","citing_title":"Large Language Models Generate Harmful Responses Using a Distinct Mechanism, Shared Across Harm Types","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07462","citing_title":"The Moltbook Files: A Harmless Slopocalypse or Humanity's Last Experiment","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17663","citing_title":"ATLAS: Constitution-Conditioned Latent Geometry and Redistribution Across Language Models and Neural Perturbation Data","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG","json":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG.json","graph_json":"https://pith.science/api/pith-number/SPCOZ5WANYK6L6LD5XPD3NC3XG/graph.json","events_json":"https://pith.science/api/pith-number/SPCOZ5WANYK6L6LD5XPD3NC3XG/events.json","paper":"https://pith.science/paper/SPCOZ5WA"},"agent_actions":{"view_html":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG","download_json":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG.json","view_paper":"https://pith.science/paper/SPCOZ5WA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11613&json=true","fetch_graph":"https://pith.science/api/pith-number/SPCOZ5WANYK6L6LD5XPD3NC3XG/graph.json","fetch_events":"https://pith.science/api/pith-number/SPCOZ5WANYK6L6LD5XPD3NC3XG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG/action/storage_attestation","attest_author":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG/action/author_attestation","sign_citation":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG/action/citation_signature","submit_replication":"https://pith.science/pith/SPCOZ5WANYK6L6LD5XPD3NC3XG/action/replication_record"}},"created_at":"2026-07-05T11:21:07.667767+00:00","updated_at":"2026-07-05T11:21:07.667767+00:00"}