{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZYTUTWYCVDYFESONR4RN5NC4NH","short_pith_number":"pith:ZYTUTWYC","schema_version":"1.0","canonical_sha256":"ce2749db02a8f05249cd8f22deb45c69eaab950c131e284a1edae8bf34fb8fd2","source":{"kind":"arxiv","id":"2403.10499","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Zero-Shot Robustness of Multimodal Foundation Models: A Pilot Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Chenguang Wang, Dawn Song, Ruoxi Jia, Xin Liu","submitted_at":"2024-03-15T17:33:49Z","abstract_excerpt":"Pre-training image representations from the raw text about images enables zero-shot vision transfer to downstream tasks. Through pre-training on millions of samples collected from the internet, multimodal foundation models, such as CLIP, produce state-of-the-art zero-shot results that often reach competitiveness with fully supervised methods without the need for task-specific training. Besides the encouraging performance on classification accuracy, it is reported that these models close the robustness gap by matching the performance of supervised models trained on ImageNet under natural distri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.10499","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-15T17:33:49Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"b310e9e9ce5b02b94d6c60e34962818db21e295d66a6b398edda8d733615e5c8","abstract_canon_sha256":"f59bd57fbc56acab471c76e8506db2b6024e36a45cbd1b784800bca305bcfa8f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:42.343968Z","signature_b64":"jot8QbtZTc8BtgzrePc/tfwpJGMZ1Tv/sJptSASS+ewiYVQGZpO8bwoHxUZN6X0FXaIXwe1W6joVlND7poAcCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce2749db02a8f05249cd8f22deb45c69eaab950c131e284a1edae8bf34fb8fd2","last_reissued_at":"2026-07-05T07:56:42.343456Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:42.343456Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Zero-Shot Robustness of Multimodal Foundation Models: A Pilot Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Chenguang Wang, Dawn Song, Ruoxi Jia, Xin Liu","submitted_at":"2024-03-15T17:33:49Z","abstract_excerpt":"Pre-training image representations from the raw text about images enables zero-shot vision transfer to downstream tasks. Through pre-training on millions of samples collected from the internet, multimodal foundation models, such as CLIP, produce state-of-the-art zero-shot results that often reach competitiveness with fully supervised methods without the need for task-specific training. Besides the encouraging performance on classification accuracy, it is reported that these models close the robustness gap by matching the performance of supervised models trained on ImageNet under natural distri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.10499","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.10499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.10499","created_at":"2026-07-05T07:56:42.343519+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.10499v1","created_at":"2026-07-05T07:56:42.343519+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.10499","created_at":"2026-07-05T07:56:42.343519+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYTUTWYCVDYF","created_at":"2026-07-05T07:56:42.343519+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYTUTWYCVDYFESON","created_at":"2026-07-05T07:56:42.343519+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYTUTWYC","created_at":"2026-07-05T07:56:42.343519+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07709","citing_title":"One Object, Multiple Lies: A Benchmark for Cross-task Adversarial Attack on Unified Vision-Language Models","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH","json":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH.json","graph_json":"https://pith.science/api/pith-number/ZYTUTWYCVDYFESONR4RN5NC4NH/graph.json","events_json":"https://pith.science/api/pith-number/ZYTUTWYCVDYFESONR4RN5NC4NH/events.json","paper":"https://pith.science/paper/ZYTUTWYC"},"agent_actions":{"view_html":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH","download_json":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH.json","view_paper":"https://pith.science/paper/ZYTUTWYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.10499&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYTUTWYCVDYFESONR4RN5NC4NH/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYTUTWYCVDYFESONR4RN5NC4NH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH/action/storage_attestation","attest_author":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH/action/author_attestation","sign_citation":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH/action/citation_signature","submit_replication":"https://pith.science/pith/ZYTUTWYCVDYFESONR4RN5NC4NH/action/replication_record"}},"created_at":"2026-07-05T07:56:42.343519+00:00","updated_at":"2026-07-05T07:56:42.343519+00:00"}