{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MBD6DYFKPFIUJENVRUK6GDVT7P","short_pith_number":"pith:MBD6DYFK","schema_version":"1.0","canonical_sha256":"6047e1e0aa79514491b58d15e30eb3fbe9832da31e41ddb615a2bd70fd022d7e","source":{"kind":"arxiv","id":"2402.02207","version":2},"attestation_state":"computed","paper":{"title":"Safety Fine-Tuning at (Almost) No Cost: A Baseline for Vision Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ondrej Bohdal, Timothy Hospedales, Tingyang Yu, Yongshuo Zong, Yongxin Yang","submitted_at":"2024-02-03T16:43:42Z","abstract_excerpt":"Current vision large language models (VLLMs) exhibit remarkable capabilities yet are prone to generate harmful content and are vulnerable to even the simplest jailbreaking attacks. Our initial analysis finds that this is due to the presence of harmful data during vision-language instruction fine-tuning, and that VLLM fine-tuning can cause forgetting of safety alignment previously learned by the underpinning LLM. To address this issue, we first curate a vision-language safe instruction-following dataset VLGuard covering various harmful categories. Our experiments demonstrate that integrating th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.02207","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-03T16:43:42Z","cross_cats_sorted":[],"title_canon_sha256":"406edc43360f8e0fc82421e1b7ca062df2800e46d32d6e3a1a56032961bdc7ce","abstract_canon_sha256":"1bd649d66ef533f6a042d345faf1a8110da4ab461c23ec1c1e3c4e73c09d5f9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:00.236455Z","signature_b64":"IrjdzenF7J9baE3TqTB5qNkRt6LE3o9/ZNpLdkaEBZRNY3puPzQiviYawYXGsGEGmKKzWVF1xX+gHbN1JQPVAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6047e1e0aa79514491b58d15e30eb3fbe9832da31e41ddb615a2bd70fd022d7e","last_reissued_at":"2026-07-05T08:33:00.235997Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:00.235997Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safety Fine-Tuning at (Almost) No Cost: A Baseline for Vision Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ondrej Bohdal, Timothy Hospedales, Tingyang Yu, Yongshuo Zong, Yongxin Yang","submitted_at":"2024-02-03T16:43:42Z","abstract_excerpt":"Current vision large language models (VLLMs) exhibit remarkable capabilities yet are prone to generate harmful content and are vulnerable to even the simplest jailbreaking attacks. Our initial analysis finds that this is due to the presence of harmful data during vision-language instruction fine-tuning, and that VLLM fine-tuning can cause forgetting of safety alignment previously learned by the underpinning LLM. To address this issue, we first curate a vision-language safe instruction-following dataset VLGuard covering various harmful categories. Our experiments demonstrate that integrating th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.02207","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.02207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.02207","created_at":"2026-07-05T08:33:00.236056+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.02207v2","created_at":"2026-07-05T08:33:00.236056+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.02207","created_at":"2026-07-05T08:33:00.236056+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBD6DYFKPFIU","created_at":"2026-07-05T08:33:00.236056+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBD6DYFKPFIUJENV","created_at":"2026-07-05T08:33:00.236056+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBD6DYFK","created_at":"2026-07-05T08:33:00.236056+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18632","citing_title":"ROBOSHACKLES: A Safety Dataset for Human-Injury Prevention in Embodied Foundation Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09125","citing_title":"Unveiling Privacy Risks in Multi-modal Large Language Models: Task-specific Vulnerabilities and Mitigation Challenges","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07970","citing_title":"Defending Against Malicious Finetuning by Scaling Train-time Adversarial Attacks","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03089","citing_title":"Constitutional On-Policy Safe Distillation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02530","citing_title":"SafeSteer: Localized On-Policy Distillation for Efficient Safety Alignment","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16403","citing_title":"When Vision Speaks for Sound","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17128","citing_title":"New Wide-Net-Casting Jailbreak Attacks Risk Large Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17310","citing_title":"Attention Hijacking: Response Manipulation Across Queries in Vision-Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11716","citing_title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2503.01743","citing_title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01449","citing_title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14219","citing_title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P","json":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P.json","graph_json":"https://pith.science/api/pith-number/MBD6DYFKPFIUJENVRUK6GDVT7P/graph.json","events_json":"https://pith.science/api/pith-number/MBD6DYFKPFIUJENVRUK6GDVT7P/events.json","paper":"https://pith.science/paper/MBD6DYFK"},"agent_actions":{"view_html":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P","download_json":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P.json","view_paper":"https://pith.science/paper/MBD6DYFK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.02207&json=true","fetch_graph":"https://pith.science/api/pith-number/MBD6DYFKPFIUJENVRUK6GDVT7P/graph.json","fetch_events":"https://pith.science/api/pith-number/MBD6DYFKPFIUJENVRUK6GDVT7P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P/action/storage_attestation","attest_author":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P/action/author_attestation","sign_citation":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P/action/citation_signature","submit_replication":"https://pith.science/pith/MBD6DYFKPFIUJENVRUK6GDVT7P/action/replication_record"}},"created_at":"2026-07-05T08:33:00.236056+00:00","updated_at":"2026-07-05T08:33:00.236056+00:00"}