{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U6KVOS4GBVK7PBFCCKIISJLCL7","short_pith_number":"pith:U6KVOS4G","schema_version":"1.0","canonical_sha256":"a795574b860d55f784a212908925625ff4b5121ea8d9887072242c72e0cd63a0","source":{"kind":"arxiv","id":"2507.03662","version":1},"attestation_state":"computed","paper":{"title":"Re-Emergent Misalignment: How Narrow Fine-Tuning Erodes Safety Alignment in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jeremiah Giordani","submitted_at":"2025-07-04T15:36:58Z","abstract_excerpt":"Recent work has shown that fine-tuning large language models (LLMs) on code with security vulnerabilities can result in misaligned and unsafe behaviors across broad domains. These results prompted concerns about the emergence of harmful behaviors from narrow domain fine-tuning. In this paper, we contextualize these findings by analyzing how such narrow adaptation impacts the internal mechanisms and behavioral manifestations of LLMs. Through a series of experiments covering output probability distributions, loss and gradient vector geometry, layer-wise activation dynamics, and activation space "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.03662","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-04T15:36:58Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a0a5c4b23326c52d4cb2937e90f7b360d6c319a365910435570fa2b13b4ee5df","abstract_canon_sha256":"e210b1c49d227bbe8d64dfe317d6d0807f72eacf0045cb1be8af4cab14e213e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:03.349408Z","signature_b64":"tpz6ifHjBDycsh1GvxmD/4KdNgqaz/jXiH94kvDvZBZ8Fm2gzPp37mT5flvFX/42vOmJDQ+qci0dhQ4Q24yfDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a795574b860d55f784a212908925625ff4b5121ea8d9887072242c72e0cd63a0","last_reissued_at":"2026-07-05T11:32:03.348987Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:03.348987Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Re-Emergent Misalignment: How Narrow Fine-Tuning Erodes Safety Alignment in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jeremiah Giordani","submitted_at":"2025-07-04T15:36:58Z","abstract_excerpt":"Recent work has shown that fine-tuning large language models (LLMs) on code with security vulnerabilities can result in misaligned and unsafe behaviors across broad domains. These results prompted concerns about the emergence of harmful behaviors from narrow domain fine-tuning. In this paper, we contextualize these findings by analyzing how such narrow adaptation impacts the internal mechanisms and behavioral manifestations of LLMs. Through a series of experiments covering output probability distributions, loss and gradient vector geometry, layer-wise activation dynamics, and activation space "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.03662","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.03662/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.03662","created_at":"2026-07-05T11:32:03.349046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.03662v1","created_at":"2026-07-05T11:32:03.349046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.03662","created_at":"2026-07-05T11:32:03.349046+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6KVOS4GBVK7","created_at":"2026-07-05T11:32:03.349046+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6KVOS4GBVK7PBFC","created_at":"2026-07-05T11:32:03.349046+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6KVOS4G","created_at":"2026-07-05T11:32:03.349046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09068","citing_title":"Emergent Misalignment Can Be Induced by Sycophancy and Reversed via Alignment Gating","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07612","citing_title":"Position: Anthropomorphic Misalignment Research Needs Stronger Evidence","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12798","citing_title":"Emergent and Subliminal Misalignment Through the Lens of Data-Mediated Transfer","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7","json":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7.json","graph_json":"https://pith.science/api/pith-number/U6KVOS4GBVK7PBFCCKIISJLCL7/graph.json","events_json":"https://pith.science/api/pith-number/U6KVOS4GBVK7PBFCCKIISJLCL7/events.json","paper":"https://pith.science/paper/U6KVOS4G"},"agent_actions":{"view_html":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7","download_json":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7.json","view_paper":"https://pith.science/paper/U6KVOS4G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.03662&json=true","fetch_graph":"https://pith.science/api/pith-number/U6KVOS4GBVK7PBFCCKIISJLCL7/graph.json","fetch_events":"https://pith.science/api/pith-number/U6KVOS4GBVK7PBFCCKIISJLCL7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7/action/storage_attestation","attest_author":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7/action/author_attestation","sign_citation":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7/action/citation_signature","submit_replication":"https://pith.science/pith/U6KVOS4GBVK7PBFCCKIISJLCL7/action/replication_record"}},"created_at":"2026-07-05T11:32:03.349046+00:00","updated_at":"2026-07-05T11:32:03.349046+00:00"}