{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VFOG632F3UXNUBV7GTLZV6LIYW","short_pith_number":"pith:VFOG632F","schema_version":"1.0","canonical_sha256":"a95c6f6f45dd2eda06bf34d79af968c5aeafea171d4646956476a24c36cc37b1","source":{"kind":"arxiv","id":"2507.16257","version":1},"attestation_state":"computed","paper":{"title":"Quality Text, Robust Vision: The Role of Language in Enhancing Visual Robustness of Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Futa Waseda, Isao Echizen, Saku Sugawara","submitted_at":"2025-07-22T06:13:30Z","abstract_excerpt":"Defending pre-trained vision-language models (VLMs), such as CLIP, against adversarial attacks is crucial, as these models are widely used in diverse zero-shot tasks, including image classification. However, existing adversarial training (AT) methods for robust fine-tuning largely overlook the role of language in enhancing visual robustness. Specifically, (1) supervised AT methods rely on short texts (e.g., class labels) to generate adversarial perturbations, leading to overfitting to object classes in the training data, and (2) unsupervised AT avoids this overfitting but remains suboptimal ag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.16257","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-22T06:13:30Z","cross_cats_sorted":[],"title_canon_sha256":"244f9c5de7df391beb802af9cd211aa31179a6f0a2a5b7b4bcaa08d8778f2f9a","abstract_canon_sha256":"7ac0c2b729f701f8806ccf8b79b453a1aa1683272814a47e52708e62f0484c2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:06.528833Z","signature_b64":"qCsToQ23+grysKCt/aFwMawDUMzyH+Eel0+uQODkyvd2udWH0YT4H91j8bO+F5/vdPvsGUo3p35PozIlukpkAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a95c6f6f45dd2eda06bf34d79af968c5aeafea171d4646956476a24c36cc37b1","last_reissued_at":"2026-07-05T11:41:06.528379Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:06.528379Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quality Text, Robust Vision: The Role of Language in Enhancing Visual Robustness of Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Futa Waseda, Isao Echizen, Saku Sugawara","submitted_at":"2025-07-22T06:13:30Z","abstract_excerpt":"Defending pre-trained vision-language models (VLMs), such as CLIP, against adversarial attacks is crucial, as these models are widely used in diverse zero-shot tasks, including image classification. However, existing adversarial training (AT) methods for robust fine-tuning largely overlook the role of language in enhancing visual robustness. Specifically, (1) supervised AT methods rely on short texts (e.g., class labels) to generate adversarial perturbations, leading to overfitting to object classes in the training data, and (2) unsupervised AT avoids this overfitting but remains suboptimal ag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.16257","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.16257/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.16257","created_at":"2026-07-05T11:41:06.528434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.16257v1","created_at":"2026-07-05T11:41:06.528434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.16257","created_at":"2026-07-05T11:41:06.528434+00:00"},{"alias_kind":"pith_short_12","alias_value":"VFOG632F3UXN","created_at":"2026-07-05T11:41:06.528434+00:00"},{"alias_kind":"pith_short_16","alias_value":"VFOG632F3UXNUBV7","created_at":"2026-07-05T11:41:06.528434+00:00"},{"alias_kind":"pith_short_8","alias_value":"VFOG632F","created_at":"2026-07-05T11:41:06.528434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW","json":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW.json","graph_json":"https://pith.science/api/pith-number/VFOG632F3UXNUBV7GTLZV6LIYW/graph.json","events_json":"https://pith.science/api/pith-number/VFOG632F3UXNUBV7GTLZV6LIYW/events.json","paper":"https://pith.science/paper/VFOG632F"},"agent_actions":{"view_html":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW","download_json":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW.json","view_paper":"https://pith.science/paper/VFOG632F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.16257&json=true","fetch_graph":"https://pith.science/api/pith-number/VFOG632F3UXNUBV7GTLZV6LIYW/graph.json","fetch_events":"https://pith.science/api/pith-number/VFOG632F3UXNUBV7GTLZV6LIYW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW/action/storage_attestation","attest_author":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW/action/author_attestation","sign_citation":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW/action/citation_signature","submit_replication":"https://pith.science/pith/VFOG632F3UXNUBV7GTLZV6LIYW/action/replication_record"}},"created_at":"2026-07-05T11:41:06.528434+00:00","updated_at":"2026-07-05T11:41:06.528434+00:00"}