{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:475ACCDMC2SD76KAJQNMO7DBKU","short_pith_number":"pith:475ACCDM","schema_version":"1.0","canonical_sha256":"e7fa01086c16a43ff9404c1ac77c61552dd0f64d6f92af015815e58c28fb5faf","source":{"kind":"arxiv","id":"2501.17433","version":1},"attestation_state":"computed","paper":{"title":"Virus: Harmful Fine-tuning Attack for Large Language Models Bypassing Guardrail Moderation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Fatih Ilhan, Ling Liu, Selim Furkan Tekin, Sihao Hu, Tiansheng Huang","submitted_at":"2025-01-29T06:24:58Z","abstract_excerpt":"Recent research shows that Large Language Models (LLMs) are vulnerable to harmful fine-tuning attacks -- models lose their safety alignment ability after fine-tuning on a few harmful samples. For risk mitigation, a guardrail is typically used to filter out harmful samples before fine-tuning. By designing a new red-teaming method, we in this paper show that purely relying on the moderation guardrail for data filtration is not reliable. Our proposed attack method, dubbed Virus, easily bypasses the guardrail moderation by slightly modifying the harmful data. Experimental results show that the har"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.17433","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-01-29T06:24:58Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"0f2af4e1672007d68b329da0e766f349e4b9cbf12dfdebf385456420fad376f8","abstract_canon_sha256":"50f1d710ebd8613ba50a47340f50d52520b03ed18cb230b3dfe2b1bd474cc824"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:52.779991Z","signature_b64":"TtPH2wUyXCgcxpS99ZAxIBBuypN8PyOi/xCUgNLC0yS2owVViFs27T6ngSTYugzYSo01lPzR5hRfFMSB6t5YBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e7fa01086c16a43ff9404c1ac77c61552dd0f64d6f92af015815e58c28fb5faf","last_reissued_at":"2026-07-05T10:06:52.779625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:52.779625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Virus: Harmful Fine-tuning Attack for Large Language Models Bypassing Guardrail Moderation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Fatih Ilhan, Ling Liu, Selim Furkan Tekin, Sihao Hu, Tiansheng Huang","submitted_at":"2025-01-29T06:24:58Z","abstract_excerpt":"Recent research shows that Large Language Models (LLMs) are vulnerable to harmful fine-tuning attacks -- models lose their safety alignment ability after fine-tuning on a few harmful samples. For risk mitigation, a guardrail is typically used to filter out harmful samples before fine-tuning. By designing a new red-teaming method, we in this paper show that purely relying on the moderation guardrail for data filtration is not reliable. Our proposed attack method, dubbed Virus, easily bypasses the guardrail moderation by slightly modifying the harmful data. Experimental results show that the har"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.17433","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.17433/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.17433","created_at":"2026-07-05T10:06:52.779684+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.17433v1","created_at":"2026-07-05T10:06:52.779684+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.17433","created_at":"2026-07-05T10:06:52.779684+00:00"},{"alias_kind":"pith_short_12","alias_value":"475ACCDMC2SD","created_at":"2026-07-05T10:06:52.779684+00:00"},{"alias_kind":"pith_short_16","alias_value":"475ACCDMC2SD76KA","created_at":"2026-07-05T10:06:52.779684+00:00"},{"alias_kind":"pith_short_8","alias_value":"475ACCDM","created_at":"2026-07-05T10:06:52.779684+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01473","citing_title":"SelfGrader: LLM Jailbreak Detection via Anchored Token-Level Logits","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU","json":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU.json","graph_json":"https://pith.science/api/pith-number/475ACCDMC2SD76KAJQNMO7DBKU/graph.json","events_json":"https://pith.science/api/pith-number/475ACCDMC2SD76KAJQNMO7DBKU/events.json","paper":"https://pith.science/paper/475ACCDM"},"agent_actions":{"view_html":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU","download_json":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU.json","view_paper":"https://pith.science/paper/475ACCDM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.17433&json=true","fetch_graph":"https://pith.science/api/pith-number/475ACCDMC2SD76KAJQNMO7DBKU/graph.json","fetch_events":"https://pith.science/api/pith-number/475ACCDMC2SD76KAJQNMO7DBKU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU/action/storage_attestation","attest_author":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU/action/author_attestation","sign_citation":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU/action/citation_signature","submit_replication":"https://pith.science/pith/475ACCDMC2SD76KAJQNMO7DBKU/action/replication_record"}},"created_at":"2026-07-05T10:06:52.779684+00:00","updated_at":"2026-07-05T10:06:52.779684+00:00"}