{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6KJVPNXMOCIAAIAUUNXNP5KEES","short_pith_number":"pith:6KJVPNXM","schema_version":"1.0","canonical_sha256":"f29357b6ec7090002014a36ed7f54424bd5bb7cd3cea1845d10583d203ec88e5","source":{"kind":"arxiv","id":"2312.12736","version":2},"attestation_state":"computed","paper":{"title":"Learning and Forgetting Unsafe Examples in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"David Madras, James Zou, Jiachen Zhao, Mengye Ren, Zhun Deng","submitted_at":"2023-12-20T03:18:50Z","abstract_excerpt":"As the number of large language models (LLMs) released to the public grows, there is a pressing need to understand the safety implications associated with these models learning from third-party custom finetuning data. We explore the behavior of LLMs finetuned on noisy custom data containing unsafe content, represented by datasets that contain biases, toxicity, and harmfulness, finding that while aligned LLMs can readily learn this unsafe content, they also tend to forget it more significantly than other examples when subsequently finetuned on safer content. Drawing inspiration from the discrep"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.12736","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-20T03:18:50Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0d61a7a9808535e0e338d9b9e854f0c3f5fe1820353f77a646c7ca14c161e5b4","abstract_canon_sha256":"c650fd9ee1f8b6e2ede7efa088cde2c368781b11610d2c9fa5751112aa498839"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:29.936488Z","signature_b64":"dS4+fc3s3ok8q02X5ssfWdScUTikP5EY5Ki8D7D7e2j6NydCHD0ODS7XcHDO56WHEYPvQeTzrhG+Y8L4PrihBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f29357b6ec7090002014a36ed7f54424bd5bb7cd3cea1845d10583d203ec88e5","last_reissued_at":"2026-07-05T08:39:29.936020Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:29.936020Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning and Forgetting Unsafe Examples in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"David Madras, James Zou, Jiachen Zhao, Mengye Ren, Zhun Deng","submitted_at":"2023-12-20T03:18:50Z","abstract_excerpt":"As the number of large language models (LLMs) released to the public grows, there is a pressing need to understand the safety implications associated with these models learning from third-party custom finetuning data. We explore the behavior of LLMs finetuned on noisy custom data containing unsafe content, represented by datasets that contain biases, toxicity, and harmfulness, finding that while aligned LLMs can readily learn this unsafe content, they also tend to forget it more significantly than other examples when subsequently finetuned on safer content. Drawing inspiration from the discrep"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.12736","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.12736/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.12736","created_at":"2026-07-05T08:39:29.936076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.12736v2","created_at":"2026-07-05T08:39:29.936076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.12736","created_at":"2026-07-05T08:39:29.936076+00:00"},{"alias_kind":"pith_short_12","alias_value":"6KJVPNXMOCIA","created_at":"2026-07-05T08:39:29.936076+00:00"},{"alias_kind":"pith_short_16","alias_value":"6KJVPNXMOCIAAIAU","created_at":"2026-07-05T08:39:29.936076+00:00"},{"alias_kind":"pith_short_8","alias_value":"6KJVPNXM","created_at":"2026-07-05T08:39:29.936076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.04992","citing_title":"You Snooze, You Lose: Automatic Safety Alignment Restoration through Neural Weight Translation","ref_index":150,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES","json":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES.json","graph_json":"https://pith.science/api/pith-number/6KJVPNXMOCIAAIAUUNXNP5KEES/graph.json","events_json":"https://pith.science/api/pith-number/6KJVPNXMOCIAAIAUUNXNP5KEES/events.json","paper":"https://pith.science/paper/6KJVPNXM"},"agent_actions":{"view_html":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES","download_json":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES.json","view_paper":"https://pith.science/paper/6KJVPNXM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.12736&json=true","fetch_graph":"https://pith.science/api/pith-number/6KJVPNXMOCIAAIAUUNXNP5KEES/graph.json","fetch_events":"https://pith.science/api/pith-number/6KJVPNXMOCIAAIAUUNXNP5KEES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES/action/storage_attestation","attest_author":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES/action/author_attestation","sign_citation":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES/action/citation_signature","submit_replication":"https://pith.science/pith/6KJVPNXMOCIAAIAUUNXNP5KEES/action/replication_record"}},"created_at":"2026-07-05T08:39:29.936076+00:00","updated_at":"2026-07-05T08:39:29.936076+00:00"}