{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XF536XGYXXSVQVWEFENLGPEI2U","short_pith_number":"pith:XF536XGY","schema_version":"1.0","canonical_sha256":"b97bbf5cd8bde55856c4291ab33c88d51132d42c4b9d3543a9506e29a4dd1183","source":{"kind":"arxiv","id":"2502.10486","version":1},"attestation_state":"computed","paper":{"title":"VLM-Guard: Safeguarding Vision-Language Models via Fulfilling Safety Alignment Gap","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CR","authors_text":"Chaowei Xiao, Fei Wang, Muhao Chen, Qin Liu","submitted_at":"2025-02-14T08:44:43Z","abstract_excerpt":"The emergence of vision language models (VLMs) comes with increased safety concerns, as the incorporation of multiple modalities heightens vulnerability to attacks. Although VLMs can be built upon LLMs that have textual safety alignment, it is easily undermined when the vision modality is integrated. We attribute this safety challenge to the modality gap, a separation of image and text in the shared representation space, which blurs the distinction between harmful and harmless queries that is evident in LLMs but weakened in VLMs. To avoid safety decay and fulfill the safety alignment gap, we p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.10486","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-02-14T08:44:43Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"9b489ba07b1c2182d22d8bdf90e9fe56e5f912e70a26f19833313b4b89df506a","abstract_canon_sha256":"f420e8859bb80908740758a698535442c6c049a45d2b1baf9ad842ad8e5cebe1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:53.454133Z","signature_b64":"kloyXIlRlN0XHdl3Xekluh+H+fcr8MIlU/FNeh9t5JIGu3kYqDBmJ4RFscqqot1eEZtdKSmH32pmuFUj7yk+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b97bbf5cd8bde55856c4291ab33c88d51132d42c4b9d3543a9506e29a4dd1183","last_reissued_at":"2026-07-05T10:14:53.453644Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:53.453644Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VLM-Guard: Safeguarding Vision-Language Models via Fulfilling Safety Alignment Gap","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CR","authors_text":"Chaowei Xiao, Fei Wang, Muhao Chen, Qin Liu","submitted_at":"2025-02-14T08:44:43Z","abstract_excerpt":"The emergence of vision language models (VLMs) comes with increased safety concerns, as the incorporation of multiple modalities heightens vulnerability to attacks. Although VLMs can be built upon LLMs that have textual safety alignment, it is easily undermined when the vision modality is integrated. We attribute this safety challenge to the modality gap, a separation of image and text in the shared representation space, which blurs the distinction between harmful and harmless queries that is evident in LLMs but weakened in VLMs. To avoid safety decay and fulfill the safety alignment gap, we p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.10486","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.10486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.10486","created_at":"2026-07-05T10:14:53.453704+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.10486v1","created_at":"2026-07-05T10:14:53.453704+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.10486","created_at":"2026-07-05T10:14:53.453704+00:00"},{"alias_kind":"pith_short_12","alias_value":"XF536XGYXXSV","created_at":"2026-07-05T10:14:53.453704+00:00"},{"alias_kind":"pith_short_16","alias_value":"XF536XGYXXSVQVWE","created_at":"2026-07-05T10:14:53.453704+00:00"},{"alias_kind":"pith_short_8","alias_value":"XF536XGY","created_at":"2026-07-05T10:14:53.453704+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15030","citing_title":"WARD: Adversarially Robust Defense of Web Agents Against Prompt Injections","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18104","citing_title":"Safety Geometry Collapse in Multimodal LLMs and Adaptive Drift Correction","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12371","citing_title":"Reading Between the Pixels: Linking Text-Image Embedding Alignment to Typographic Attack Success on Vision-Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06247","citing_title":"SALLIE: Generation-Free Hidden-State Detection of Jailbreaks and Prompt Injections Across Text and Vision","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U","json":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U.json","graph_json":"https://pith.science/api/pith-number/XF536XGYXXSVQVWEFENLGPEI2U/graph.json","events_json":"https://pith.science/api/pith-number/XF536XGYXXSVQVWEFENLGPEI2U/events.json","paper":"https://pith.science/paper/XF536XGY"},"agent_actions":{"view_html":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U","download_json":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U.json","view_paper":"https://pith.science/paper/XF536XGY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.10486&json=true","fetch_graph":"https://pith.science/api/pith-number/XF536XGYXXSVQVWEFENLGPEI2U/graph.json","fetch_events":"https://pith.science/api/pith-number/XF536XGYXXSVQVWEFENLGPEI2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U/action/storage_attestation","attest_author":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U/action/author_attestation","sign_citation":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U/action/citation_signature","submit_replication":"https://pith.science/pith/XF536XGYXXSVQVWEFENLGPEI2U/action/replication_record"}},"created_at":"2026-07-05T10:14:53.453704+00:00","updated_at":"2026-07-05T10:14:53.453704+00:00"}