{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VPIPNJXDXQ674WLE2DVZZVTYXJ","short_pith_number":"pith:VPIPNJXD","schema_version":"1.0","canonical_sha256":"abd0f6a6e3bc3dfe5964d0eb9cd678ba74a21fb2f59f3a5b248333d1cd76ca9f","source":{"kind":"arxiv","id":"2405.10529","version":2},"attestation_state":"computed","paper":{"title":"Safeguarding Vision-Language Models Against Patched Visual Prompt Injectors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Changsheng Wang, Chaowei Xiao, Jiachen Sun, Jiongxiao Wang, Yiwei Zhang","submitted_at":"2024-05-17T04:19:19Z","abstract_excerpt":"Large language models have become increasingly prominent, also signaling a shift towards multimodality as the next frontier in artificial intelligence, where their embeddings are harnessed as prompts to generate textual content. Vision-language models (VLMs) stand at the forefront of this advancement, offering innovative ways to combine visual and textual data for enhanced understanding and interaction. However, this integration also enlarges the attack surface. Patch-based adversarial attack is considered the most realistic threat model in physical vision applications, as demonstrated in many"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.10529","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-17T04:19:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"63447ae4dc6191e974a3965db4d5a5086105f872618daf68f9cada400069d512","abstract_canon_sha256":"cd4d8d8cdb4e898090f71ca2fd51250a6b373ff8dffa78111a8e40bdb67b0759"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:44.353721Z","signature_b64":"SCrhBFhAr1Mg5FC4PAGI0JmZCmPHpuJLoBVvLxX8M+Kt9YEaWoqD/GDnLjaeRsbUOb8eX10XvCHmyX5gIM78DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abd0f6a6e3bc3dfe5964d0eb9cd678ba74a21fb2f59f3a5b248333d1cd76ca9f","last_reissued_at":"2026-07-05T08:58:44.353239Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:44.353239Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safeguarding Vision-Language Models Against Patched Visual Prompt Injectors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Changsheng Wang, Chaowei Xiao, Jiachen Sun, Jiongxiao Wang, Yiwei Zhang","submitted_at":"2024-05-17T04:19:19Z","abstract_excerpt":"Large language models have become increasingly prominent, also signaling a shift towards multimodality as the next frontier in artificial intelligence, where their embeddings are harnessed as prompts to generate textual content. Vision-language models (VLMs) stand at the forefront of this advancement, offering innovative ways to combine visual and textual data for enhanced understanding and interaction. However, this integration also enlarges the attack surface. Patch-based adversarial attack is considered the most realistic threat model in physical vision applications, as demonstrated in many"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.10529","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.10529/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.10529","created_at":"2026-07-05T08:58:44.353297+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.10529v2","created_at":"2026-07-05T08:58:44.353297+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.10529","created_at":"2026-07-05T08:58:44.353297+00:00"},{"alias_kind":"pith_short_12","alias_value":"VPIPNJXDXQ67","created_at":"2026-07-05T08:58:44.353297+00:00"},{"alias_kind":"pith_short_16","alias_value":"VPIPNJXDXQ674WLE","created_at":"2026-07-05T08:58:44.353297+00:00"},{"alias_kind":"pith_short_8","alias_value":"VPIPNJXD","created_at":"2026-07-05T08:58:44.353297+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10904","citing_title":"Auditing Inference-Time Defense Evaluation for Multimodal Large Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09125","citing_title":"Unveiling Privacy Risks in Multi-modal Large Language Models: Task-specific Vulnerabilities and Mitigation Challenges","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16090","citing_title":"A Cross-Modal Prompt Injection Attack against Large Vision-Language Models with Image-Only Perturbation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03995","citing_title":"A Systematic Study of Cross-Modal Typographic Attacks on Audio-Visual Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25562","citing_title":"SnapGuard: Lightweight Prompt Injection Detection for Screenshot-Based Web Agents","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ","json":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ.json","graph_json":"https://pith.science/api/pith-number/VPIPNJXDXQ674WLE2DVZZVTYXJ/graph.json","events_json":"https://pith.science/api/pith-number/VPIPNJXDXQ674WLE2DVZZVTYXJ/events.json","paper":"https://pith.science/paper/VPIPNJXD"},"agent_actions":{"view_html":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ","download_json":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ.json","view_paper":"https://pith.science/paper/VPIPNJXD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.10529&json=true","fetch_graph":"https://pith.science/api/pith-number/VPIPNJXDXQ674WLE2DVZZVTYXJ/graph.json","fetch_events":"https://pith.science/api/pith-number/VPIPNJXDXQ674WLE2DVZZVTYXJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ/action/storage_attestation","attest_author":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ/action/author_attestation","sign_citation":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ/action/citation_signature","submit_replication":"https://pith.science/pith/VPIPNJXDXQ674WLE2DVZZVTYXJ/action/replication_record"}},"created_at":"2026-07-05T08:58:44.353297+00:00","updated_at":"2026-07-05T08:58:44.353297+00:00"}