{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y3CHDKXR2VWM4AM5QHWC4QSDID","short_pith_number":"pith:Y3CHDKXR","schema_version":"1.0","canonical_sha256":"c6c471aaf1d56cce019d81ec2e424340c20ce3f9b61baab4050b0f7f4e3f18b5","source":{"kind":"arxiv","id":"2411.13136","version":1},"attestation_state":"computed","paper":{"title":"TAPT: Test-Time Adversarial Prompt Tuning for Robust Inference in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiaming Zhang, Jingjing Chen, Kai Chen, Xingjun Ma, Xin Wang","submitted_at":"2024-11-20T08:58:59Z","abstract_excerpt":"Large pre-trained Vision-Language Models (VLMs) such as CLIP have demonstrated excellent zero-shot generalizability across various downstream tasks. However, recent studies have shown that the inference performance of CLIP can be greatly degraded by small adversarial perturbations, especially its visual modality, posing significant safety threats. To mitigate this vulnerability, in this paper, we propose a novel defense method called Test-Time Adversarial Prompt Tuning (TAPT) to enhance the inference robustness of CLIP against visual adversarial attacks. TAPT is a test-time defense method that"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.13136","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-20T08:58:59Z","cross_cats_sorted":[],"title_canon_sha256":"49376ac4ee9f1f5615bf82968e1811051c67aca7d2548425e335429a6fde7135","abstract_canon_sha256":"6505e628b2db0021f1dc8914883dedd5eb5fb7796370cc871936ce3318caf6b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:05.591549Z","signature_b64":"sYmKdaoXUBMo8a6bBJ0dL1Mjvi5wrANwA+fvRiNpvfzdT2kPAowD9TBzFGmpBqtox+IPAWA7SWQGnXoOz4i2Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c6c471aaf1d56cce019d81ec2e424340c20ce3f9b61baab4050b0f7f4e3f18b5","last_reissued_at":"2026-07-05T09:38:05.591144Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:05.591144Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TAPT: Test-Time Adversarial Prompt Tuning for Robust Inference in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiaming Zhang, Jingjing Chen, Kai Chen, Xingjun Ma, Xin Wang","submitted_at":"2024-11-20T08:58:59Z","abstract_excerpt":"Large pre-trained Vision-Language Models (VLMs) such as CLIP have demonstrated excellent zero-shot generalizability across various downstream tasks. However, recent studies have shown that the inference performance of CLIP can be greatly degraded by small adversarial perturbations, especially its visual modality, posing significant safety threats. To mitigate this vulnerability, in this paper, we propose a novel defense method called Test-Time Adversarial Prompt Tuning (TAPT) to enhance the inference robustness of CLIP against visual adversarial attacks. TAPT is a test-time defense method that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.13136","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.13136/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.13136","created_at":"2026-07-05T09:38:05.591206+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.13136v1","created_at":"2026-07-05T09:38:05.591206+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.13136","created_at":"2026-07-05T09:38:05.591206+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y3CHDKXR2VWM","created_at":"2026-07-05T09:38:05.591206+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y3CHDKXR2VWM4AM5","created_at":"2026-07-05T09:38:05.591206+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y3CHDKXR","created_at":"2026-07-05T09:38:05.591206+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID","json":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID.json","graph_json":"https://pith.science/api/pith-number/Y3CHDKXR2VWM4AM5QHWC4QSDID/graph.json","events_json":"https://pith.science/api/pith-number/Y3CHDKXR2VWM4AM5QHWC4QSDID/events.json","paper":"https://pith.science/paper/Y3CHDKXR"},"agent_actions":{"view_html":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID","download_json":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID.json","view_paper":"https://pith.science/paper/Y3CHDKXR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.13136&json=true","fetch_graph":"https://pith.science/api/pith-number/Y3CHDKXR2VWM4AM5QHWC4QSDID/graph.json","fetch_events":"https://pith.science/api/pith-number/Y3CHDKXR2VWM4AM5QHWC4QSDID/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID/action/storage_attestation","attest_author":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID/action/author_attestation","sign_citation":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID/action/citation_signature","submit_replication":"https://pith.science/pith/Y3CHDKXR2VWM4AM5QHWC4QSDID/action/replication_record"}},"created_at":"2026-07-05T09:38:05.591206+00:00","updated_at":"2026-07-05T09:38:05.591206+00:00"}