{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6KDF2YFPYFHAYNXFEIDO7FWDO5","short_pith_number":"pith:6KDF2YFP","schema_version":"1.0","canonical_sha256":"f2865d60afc14e0c36e52206ef96c377471ac20336d78d7a7ba9a034a0a05d1b","source":{"kind":"arxiv","id":"2210.09545","version":1},"attestation_state":"computed","paper":{"title":"Fine-mixing: Mitigating Backdoors in Fine-tuned Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenguang Wang, Lingjuan Lyu, Xingjun Ma, Xu Sun, Zhiyuan Zhang","submitted_at":"2022-10-18T02:44:38Z","abstract_excerpt":"Deep Neural Networks (DNNs) are known to be vulnerable to backdoor attacks. In Natural Language Processing (NLP), DNNs are often backdoored during the fine-tuning process of a large-scale Pre-trained Language Model (PLM) with poisoned samples. Although the clean weights of PLMs are readily available, existing methods have ignored this information in defending NLP models against backdoor attacks. In this work, we take the first step to exploit the pre-trained (unfine-tuned) weights to mitigate backdoors in fine-tuned language models. Specifically, we leverage the clean pre-trained weights via t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.09545","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-18T02:44:38Z","cross_cats_sorted":["cs.CR","cs.LG"],"title_canon_sha256":"0db33a9fbd06333f68b051251ea0d859c112b2ed6057a5079c07095ba7edb50f","abstract_canon_sha256":"b1e5da1375e2ff3d1a57c63028ad1ecbd24193160d2e994dad3454c6b4c43518"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:07:56.669657Z","signature_b64":"WPIuatBsecSNA5BV5952V4Njr1kP8RmE83RCXaIywMKmquOfey59mdFxRpXZKcwOJ137ThDRpY2MZ1IOtJZKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2865d60afc14e0c36e52206ef96c377471ac20336d78d7a7ba9a034a0a05d1b","last_reissued_at":"2026-07-05T05:07:56.669157Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:07:56.669157Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-mixing: Mitigating Backdoors in Fine-tuned Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenguang Wang, Lingjuan Lyu, Xingjun Ma, Xu Sun, Zhiyuan Zhang","submitted_at":"2022-10-18T02:44:38Z","abstract_excerpt":"Deep Neural Networks (DNNs) are known to be vulnerable to backdoor attacks. In Natural Language Processing (NLP), DNNs are often backdoored during the fine-tuning process of a large-scale Pre-trained Language Model (PLM) with poisoned samples. Although the clean weights of PLMs are readily available, existing methods have ignored this information in defending NLP models against backdoor attacks. In this work, we take the first step to exploit the pre-trained (unfine-tuned) weights to mitigate backdoors in fine-tuned language models. Specifically, we leverage the clean pre-trained weights via t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.09545","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.09545/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.09545","created_at":"2026-07-05T05:07:56.669216+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.09545v1","created_at":"2026-07-05T05:07:56.669216+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.09545","created_at":"2026-07-05T05:07:56.669216+00:00"},{"alias_kind":"pith_short_12","alias_value":"6KDF2YFPYFHA","created_at":"2026-07-05T05:07:56.669216+00:00"},{"alias_kind":"pith_short_16","alias_value":"6KDF2YFPYFHAYNXF","created_at":"2026-07-05T05:07:56.669216+00:00"},{"alias_kind":"pith_short_8","alias_value":"6KDF2YFP","created_at":"2026-07-05T05:07:56.669216+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29239","citing_title":"Breaking the Rounding Trap: Securing LLMs against Quantization-Conditioned Backdoors","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13789","citing_title":"Uncovering and Aligning Anomalous Attention Heads to Defend Against NLP Backdoor Attacks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10253","citing_title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12529","citing_title":"BackFlush: Knowledge-Free Backdoor Detection and Elimination with Watermark Preservation in Large Language Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5","json":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5.json","graph_json":"https://pith.science/api/pith-number/6KDF2YFPYFHAYNXFEIDO7FWDO5/graph.json","events_json":"https://pith.science/api/pith-number/6KDF2YFPYFHAYNXFEIDO7FWDO5/events.json","paper":"https://pith.science/paper/6KDF2YFP"},"agent_actions":{"view_html":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5","download_json":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5.json","view_paper":"https://pith.science/paper/6KDF2YFP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.09545&json=true","fetch_graph":"https://pith.science/api/pith-number/6KDF2YFPYFHAYNXFEIDO7FWDO5/graph.json","fetch_events":"https://pith.science/api/pith-number/6KDF2YFPYFHAYNXFEIDO7FWDO5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5/action/storage_attestation","attest_author":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5/action/author_attestation","sign_citation":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5/action/citation_signature","submit_replication":"https://pith.science/pith/6KDF2YFPYFHAYNXFEIDO7FWDO5/action/replication_record"}},"created_at":"2026-07-05T05:07:56.669216+00:00","updated_at":"2026-07-05T05:07:56.669216+00:00"}