{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5FIJNWKYW6UFTFJ3HILVDVLDZM","short_pith_number":"pith:5FIJNWKY","schema_version":"1.0","canonical_sha256":"e95096d958b7a859953b3a1751d563cb1647a6ef0b9bae72d824da2c6399b8fe","source":{"kind":"arxiv","id":"2305.16934","version":2},"attestation_state":"computed","paper":{"title":"On Evaluating Adversarial Robustness of Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Chao Du, Chongxuan Li, Min Lin, Ngai-Man Cheung, Tianyu Pang, Xiao Yang, Yunqing Zhao","submitted_at":"2023-05-26T13:49:44Z","abstract_excerpt":"Large vision-language models (VLMs) such as GPT-4 have achieved unprecedented performance in response generation, especially with visual inputs, enabling more creative and adaptable interaction than large language models such as ChatGPT. Nonetheless, multimodal generation exacerbates safety concerns, since adversaries may successfully evade the entire system by subtly manipulating the most vulnerable modality (e.g., vision). To this end, we propose evaluating the robustness of open-source large VLMs in the most realistic and high-risk setting, where adversaries have only black-box system acces"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16934","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-26T13:49:44Z","cross_cats_sorted":["cs.CL","cs.CR","cs.LG","cs.MM"],"title_canon_sha256":"4dc0bab33468e05295eeceaf345e927f4ac1245708a917121711141153cac56d","abstract_canon_sha256":"4b398d694a36f2ee536090cd5b7c3a923a6eeb9f9fc0268ccf0b20bb6054a471"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:28.089520Z","signature_b64":"kmKYBmiBcXij0LzagQrUFCjOSXUd81s4WijXHEdp6rKjxeiIhRDokSCIepJBkdK2QRlYEVM/aUuMhpY3AoS5BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e95096d958b7a859953b3a1751d563cb1647a6ef0b9bae72d824da2c6399b8fe","last_reissued_at":"2026-07-05T07:06:28.089023Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:28.089023Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Evaluating Adversarial Robustness of Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Chao Du, Chongxuan Li, Min Lin, Ngai-Man Cheung, Tianyu Pang, Xiao Yang, Yunqing Zhao","submitted_at":"2023-05-26T13:49:44Z","abstract_excerpt":"Large vision-language models (VLMs) such as GPT-4 have achieved unprecedented performance in response generation, especially with visual inputs, enabling more creative and adaptable interaction than large language models such as ChatGPT. Nonetheless, multimodal generation exacerbates safety concerns, since adversaries may successfully evade the entire system by subtly manipulating the most vulnerable modality (e.g., vision). To this end, we propose evaluating the robustness of open-source large VLMs in the most realistic and high-risk setting, where adversaries have only black-box system acces"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16934","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16934","created_at":"2026-07-05T07:06:28.089079+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16934v2","created_at":"2026-07-05T07:06:28.089079+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16934","created_at":"2026-07-05T07:06:28.089079+00:00"},{"alias_kind":"pith_short_12","alias_value":"5FIJNWKYW6UF","created_at":"2026-07-05T07:06:28.089079+00:00"},{"alias_kind":"pith_short_16","alias_value":"5FIJNWKYW6UFTFJ3","created_at":"2026-07-05T07:06:28.089079+00:00"},{"alias_kind":"pith_short_8","alias_value":"5FIJNWKY","created_at":"2026-07-05T07:06:28.089079+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26566","citing_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06964","citing_title":"Adversarial Video Promotion Against Text-to-Video Retrieval","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16502","citing_title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03744","citing_title":"Improved Baselines with Visual Instruction Tuning","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13394","citing_title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13803","citing_title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM","json":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM.json","graph_json":"https://pith.science/api/pith-number/5FIJNWKYW6UFTFJ3HILVDVLDZM/graph.json","events_json":"https://pith.science/api/pith-number/5FIJNWKYW6UFTFJ3HILVDVLDZM/events.json","paper":"https://pith.science/paper/5FIJNWKY"},"agent_actions":{"view_html":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM","download_json":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM.json","view_paper":"https://pith.science/paper/5FIJNWKY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16934&json=true","fetch_graph":"https://pith.science/api/pith-number/5FIJNWKYW6UFTFJ3HILVDVLDZM/graph.json","fetch_events":"https://pith.science/api/pith-number/5FIJNWKYW6UFTFJ3HILVDVLDZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM/action/storage_attestation","attest_author":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM/action/author_attestation","sign_citation":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM/action/citation_signature","submit_replication":"https://pith.science/pith/5FIJNWKYW6UFTFJ3HILVDVLDZM/action/replication_record"}},"created_at":"2026-07-05T07:06:28.089079+00:00","updated_at":"2026-07-05T07:06:28.089079+00:00"}