{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:S6CT7TRMY5VTLEKX4M5GENWCX5","short_pith_number":"pith:S6CT7TRM","schema_version":"1.0","canonical_sha256":"97853fce2cc76b359157e33a6236c2bf6676488f5a69696679a9f3df3144e7d1","source":{"kind":"arxiv","id":"2501.16378","version":1},"attestation_state":"computed","paper":{"title":"Internal Activation Revision: Safeguarding Vision Language Models Without Parameter Update","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Fakhri Karray, Jiahui Geng, Kun Song, Lei Ma, Qing Li, Zongxiong Chen","submitted_at":"2025-01-24T06:17:22Z","abstract_excerpt":"Vision-language models (VLMs) demonstrate strong multimodal capabilities but have been found to be more susceptible to generating harmful content compared to their backbone large language models (LLMs). Our investigation reveals that the integration of images significantly shifts the model's internal activations during the forward pass, diverging from those triggered by textual input. Moreover, the safety alignments of LLMs embedded within VLMs are not sufficiently robust to handle the activations discrepancies, making the models vulnerable to even the simplest jailbreaking attacks. To address"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16378","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-24T06:17:22Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"a61f05c1a372f81aa58da65d888291da5a2f69b763014a35b7f4ad28fcfe75e6","abstract_canon_sha256":"eb865d1a3e23ed6dd5f717bbddeab3ba382495362bd741d308b9b2f7f111e11e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:12.081191Z","signature_b64":"KVMDuXeF154RE2uG/iuJZhH+uIYBU78R+qsQekRcuyy3mtaPUK3XY1Dqxo0YnRec7AGbJTAiNpY02IIXgI8mDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"97853fce2cc76b359157e33a6236c2bf6676488f5a69696679a9f3df3144e7d1","last_reissued_at":"2026-07-05T10:06:12.080749Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:12.080749Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Internal Activation Revision: Safeguarding Vision Language Models Without Parameter Update","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Fakhri Karray, Jiahui Geng, Kun Song, Lei Ma, Qing Li, Zongxiong Chen","submitted_at":"2025-01-24T06:17:22Z","abstract_excerpt":"Vision-language models (VLMs) demonstrate strong multimodal capabilities but have been found to be more susceptible to generating harmful content compared to their backbone large language models (LLMs). Our investigation reveals that the integration of images significantly shifts the model's internal activations during the forward pass, diverging from those triggered by textual input. Moreover, the safety alignments of LLMs embedded within VLMs are not sufficiently robust to handle the activations discrepancies, making the models vulnerable to even the simplest jailbreaking attacks. To address"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16378","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16378","created_at":"2026-07-05T10:06:12.080809+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16378v1","created_at":"2026-07-05T10:06:12.080809+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16378","created_at":"2026-07-05T10:06:12.080809+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6CT7TRMY5VT","created_at":"2026-07-05T10:06:12.080809+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6CT7TRMY5VTLEKX","created_at":"2026-07-05T10:06:12.080809+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6CT7TRM","created_at":"2026-07-05T10:06:12.080809+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.22037","citing_title":"Secure Tug-of-War (SecTOW): Iterative Defense-Attack Training with Reinforcement Learning for Multimodal Model Security","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5","json":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5.json","graph_json":"https://pith.science/api/pith-number/S6CT7TRMY5VTLEKX4M5GENWCX5/graph.json","events_json":"https://pith.science/api/pith-number/S6CT7TRMY5VTLEKX4M5GENWCX5/events.json","paper":"https://pith.science/paper/S6CT7TRM"},"agent_actions":{"view_html":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5","download_json":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5.json","view_paper":"https://pith.science/paper/S6CT7TRM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16378&json=true","fetch_graph":"https://pith.science/api/pith-number/S6CT7TRMY5VTLEKX4M5GENWCX5/graph.json","fetch_events":"https://pith.science/api/pith-number/S6CT7TRMY5VTLEKX4M5GENWCX5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5/action/storage_attestation","attest_author":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5/action/author_attestation","sign_citation":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5/action/citation_signature","submit_replication":"https://pith.science/pith/S6CT7TRMY5VTLEKX4M5GENWCX5/action/replication_record"}},"created_at":"2026-07-05T10:06:12.080809+00:00","updated_at":"2026-07-05T10:06:12.080809+00:00"}