{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VXRHX2ZNU3BUZEIHLWDMHJ6HEZ","short_pith_number":"pith:VXRHX2ZN","schema_version":"1.0","canonical_sha256":"ade27beb2da6c34c91075d86c3a7c72673bf8c91ee1ca04796dbc88851106d36","source":{"kind":"arxiv","id":"2308.12636","version":5},"attestation_state":"computed","paper":{"title":"Exploring Transferability of Multimodal Adversarial Samples for Vision-Language Pre-training Models with Contrastive Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.MM","authors_text":"Hang Su, Hanwang Zhang, Richang Hong, Wenbo Hu, Yinpeng Dong, Youze Wang","submitted_at":"2023-08-24T08:22:21Z","abstract_excerpt":"The integration of visual and textual data in Vision-Language Pre-training (VLP) models is crucial for enhancing vision-language understanding. However, the adversarial robustness of these models, especially in the alignment of image-text features, has not yet been sufficiently explored. In this paper, we introduce a novel gradient-based multimodal adversarial attack method, underpinned by contrastive learning, to improve the transferability of multimodal adversarial samples in VLP models. This method concurrently generates adversarial texts and images within imperceptive perturbation, employi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.12636","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2023-08-24T08:22:21Z","cross_cats_sorted":[],"title_canon_sha256":"bd1806267730496b22d18ae50d495c8830c237ad6dca72334a47c94429f715bb","abstract_canon_sha256":"d026a8da87a416723c5fbed238523e20f19c20d586ed036adb06787bb3b98823"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:54.083598Z","signature_b64":"oDKGQ15jncLEs/FoWdzSBBgM3qH/RFRyPrvji8LhUFMAvHYZeja+h6GPJVK/QCd2dyh0m752/12bkkXLSlHbBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ade27beb2da6c34c91075d86c3a7c72673bf8c91ee1ca04796dbc88851106d36","last_reissued_at":"2026-07-05T11:13:54.083174Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:54.083174Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Transferability of Multimodal Adversarial Samples for Vision-Language Pre-training Models with Contrastive Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.MM","authors_text":"Hang Su, Hanwang Zhang, Richang Hong, Wenbo Hu, Yinpeng Dong, Youze Wang","submitted_at":"2023-08-24T08:22:21Z","abstract_excerpt":"The integration of visual and textual data in Vision-Language Pre-training (VLP) models is crucial for enhancing vision-language understanding. However, the adversarial robustness of these models, especially in the alignment of image-text features, has not yet been sufficiently explored. In this paper, we introduce a novel gradient-based multimodal adversarial attack method, underpinned by contrastive learning, to improve the transferability of multimodal adversarial samples in VLP models. This method concurrently generates adversarial texts and images within imperceptive perturbation, employi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.12636","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.12636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.12636","created_at":"2026-07-05T11:13:54.083229+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.12636v5","created_at":"2026-07-05T11:13:54.083229+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.12636","created_at":"2026-07-05T11:13:54.083229+00:00"},{"alias_kind":"pith_short_12","alias_value":"VXRHX2ZNU3BU","created_at":"2026-07-05T11:13:54.083229+00:00"},{"alias_kind":"pith_short_16","alias_value":"VXRHX2ZNU3BUZEIH","created_at":"2026-07-05T11:13:54.083229+00:00"},{"alias_kind":"pith_short_8","alias_value":"VXRHX2ZN","created_at":"2026-07-05T11:13:54.083229+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":224,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17577","citing_title":"TAME: Test-Time Adversarial Prompt Tuning via Mixture-of-Experts for Vision-Language Models","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ","json":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ.json","graph_json":"https://pith.science/api/pith-number/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/graph.json","events_json":"https://pith.science/api/pith-number/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/events.json","paper":"https://pith.science/paper/VXRHX2ZN"},"agent_actions":{"view_html":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ","download_json":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ.json","view_paper":"https://pith.science/paper/VXRHX2ZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.12636&json=true","fetch_graph":"https://pith.science/api/pith-number/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/graph.json","fetch_events":"https://pith.science/api/pith-number/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/action/storage_attestation","attest_author":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/action/author_attestation","sign_citation":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/action/citation_signature","submit_replication":"https://pith.science/pith/VXRHX2ZNU3BUZEIHLWDMHJ6HEZ/action/replication_record"}},"created_at":"2026-07-05T11:13:54.083229+00:00","updated_at":"2026-07-05T11:13:54.083229+00:00"}