{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EROAGULYYGRY4ERNU4X4WN6IDY","short_pith_number":"pith:EROAGULY","schema_version":"1.0","canonical_sha256":"245c035178c1a38e122da72fcb37c81e0d5b37903248821b5e6a49e3c201dfdc","source":{"kind":"arxiv","id":"2404.06244","version":1},"attestation_state":"computed","paper":{"title":"Anchor-based Robust Finetuning of Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gui-Song Xia, Jinwei Han, Ke Yan, Shouhong Ding, Yingguo Gao, Yuan Gao, Zhiwen Lin, Zhongyisun Sun","submitted_at":"2024-04-09T12:10:54Z","abstract_excerpt":"We aim at finetuning a vision-language model without hurting its out-of-distribution (OOD) generalization. We address two types of OOD generalization, i.e., i) domain shift such as natural to sketch images, and ii) zero-shot capability to recognize the category that was not contained in the finetune data. Arguably, the diminished OOD generalization after finetuning stems from the excessively simplified finetuning target, which only provides the class information, such as ``a photo of a [CLASS]''. This is distinct from the process in that CLIP was pretrained, where there is abundant text superv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.06244","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-09T12:10:54Z","cross_cats_sorted":[],"title_canon_sha256":"64ea3c50519dc3b6e997c0de470b908b5c9ebe6ed14180805a26b0fc477ffd79","abstract_canon_sha256":"b94b6b117e0c42e3d652aa008cc1bee26e90fe6973d2474018ea7cdc15cabcb7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:06.451476Z","signature_b64":"zNSO731yxp2yTipfeHGp85hv2YNpYfFDrwr8JjoRSb6YLDYnQEzSnb3Rhx/+HOje0IhKTbCPHRFUxw5Tx+IAAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"245c035178c1a38e122da72fcb37c81e0d5b37903248821b5e6a49e3c201dfdc","last_reissued_at":"2026-07-05T08:06:06.451008Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:06.451008Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Anchor-based Robust Finetuning of Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gui-Song Xia, Jinwei Han, Ke Yan, Shouhong Ding, Yingguo Gao, Yuan Gao, Zhiwen Lin, Zhongyisun Sun","submitted_at":"2024-04-09T12:10:54Z","abstract_excerpt":"We aim at finetuning a vision-language model without hurting its out-of-distribution (OOD) generalization. We address two types of OOD generalization, i.e., i) domain shift such as natural to sketch images, and ii) zero-shot capability to recognize the category that was not contained in the finetune data. Arguably, the diminished OOD generalization after finetuning stems from the excessively simplified finetuning target, which only provides the class information, such as ``a photo of a [CLASS]''. This is distinct from the process in that CLIP was pretrained, where there is abundant text superv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.06244","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.06244/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.06244","created_at":"2026-07-05T08:06:06.451068+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.06244v1","created_at":"2026-07-05T08:06:06.451068+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.06244","created_at":"2026-07-05T08:06:06.451068+00:00"},{"alias_kind":"pith_short_12","alias_value":"EROAGULYYGRY","created_at":"2026-07-05T08:06:06.451068+00:00"},{"alias_kind":"pith_short_16","alias_value":"EROAGULYYGRY4ERN","created_at":"2026-07-05T08:06:06.451068+00:00"},{"alias_kind":"pith_short_8","alias_value":"EROAGULY","created_at":"2026-07-05T08:06:06.451068+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.14100","citing_title":"A Hierarchical Test Platform for Vision Language Model (VLM)-Integrated Real-World Autonomous Driving","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY","json":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY.json","graph_json":"https://pith.science/api/pith-number/EROAGULYYGRY4ERNU4X4WN6IDY/graph.json","events_json":"https://pith.science/api/pith-number/EROAGULYYGRY4ERNU4X4WN6IDY/events.json","paper":"https://pith.science/paper/EROAGULY"},"agent_actions":{"view_html":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY","download_json":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY.json","view_paper":"https://pith.science/paper/EROAGULY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.06244&json=true","fetch_graph":"https://pith.science/api/pith-number/EROAGULYYGRY4ERNU4X4WN6IDY/graph.json","fetch_events":"https://pith.science/api/pith-number/EROAGULYYGRY4ERNU4X4WN6IDY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY/action/storage_attestation","attest_author":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY/action/author_attestation","sign_citation":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY/action/citation_signature","submit_replication":"https://pith.science/pith/EROAGULYYGRY4ERNU4X4WN6IDY/action/replication_record"}},"created_at":"2026-07-05T08:06:06.451068+00:00","updated_at":"2026-07-05T08:06:06.451068+00:00"}