{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JH6WLZO343ISMPDAIVEMG6MSPE","short_pith_number":"pith:JH6WLZO3","schema_version":"1.0","canonical_sha256":"49fd65e5dbe6d1263c604548c37992791aeb6fb48b5502255df831ea6a8f7d3e","source":{"kind":"arxiv","id":"2305.18010","version":2},"attestation_state":"computed","paper":{"title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Linchao Zhu, Shuai Zhao, Xiaohan Wang, Yi Yang","submitted_at":"2023-05-29T11:03:59Z","abstract_excerpt":"One fascinating aspect of pre-trained vision-language models~(VLMs) learning under language supervision is their impressive zero-shot generalization capability. However, this ability is hindered by distribution shifts between the training and testing data. Previous test time adaptation~(TTA) methods for VLMs in zero-shot classification rely on minimizing the entropy of model outputs, tending to be stuck in incorrect model predictions. In this work, we propose TTA with feedback to rectify the model output and prevent the model from becoming blindly confident. Specifically, a CLIP model is adopt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18010","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-05-29T11:03:59Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"e78c936bff2033039e31ada4f2a06ee5ffc2c62d234b3c76a6454778add7d4cd","abstract_canon_sha256":"a98f18bcf33cecd3de763b5eb7273be168903586efe5f2e70dc33f7170026946"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:25.253552Z","signature_b64":"2W4xoYsgN/OxtbEbKbGUFDFzOm0RIYhr7oLce3QxxTuOQhfe9IrCxqr3FIr+2F7OUc8C5oGMyfadE41DiIn3DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49fd65e5dbe6d1263c604548c37992791aeb6fb48b5502255df831ea6a8f7d3e","last_reissued_at":"2026-07-05T07:47:25.253152Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:25.253152Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Linchao Zhu, Shuai Zhao, Xiaohan Wang, Yi Yang","submitted_at":"2023-05-29T11:03:59Z","abstract_excerpt":"One fascinating aspect of pre-trained vision-language models~(VLMs) learning under language supervision is their impressive zero-shot generalization capability. However, this ability is hindered by distribution shifts between the training and testing data. Previous test time adaptation~(TTA) methods for VLMs in zero-shot classification rely on minimizing the entropy of model outputs, tending to be stuck in incorrect model predictions. In this work, we propose TTA with feedback to rectify the model output and prevent the model from becoming blindly confident. Specifically, a CLIP model is adopt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18010","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18010/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18010","created_at":"2026-07-05T07:47:25.253215+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18010v2","created_at":"2026-07-05T07:47:25.253215+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18010","created_at":"2026-07-05T07:47:25.253215+00:00"},{"alias_kind":"pith_short_12","alias_value":"JH6WLZO343IS","created_at":"2026-07-05T07:47:25.253215+00:00"},{"alias_kind":"pith_short_16","alias_value":"JH6WLZO343ISMPDA","created_at":"2026-07-05T07:47:25.253215+00:00"},{"alias_kind":"pith_short_8","alias_value":"JH6WLZO3","created_at":"2026-07-05T07:47:25.253215+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23257","citing_title":"Turning Adaptation into Assets: Cross-Domain Bridging for Online Vision-Language Navigation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26221","citing_title":"Seeking Consensus: Geometric-Semantic On-the-Fly Recalibration for Open-Vocabulary Remote Sensing Semantic Segmentation","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE","json":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE.json","graph_json":"https://pith.science/api/pith-number/JH6WLZO343ISMPDAIVEMG6MSPE/graph.json","events_json":"https://pith.science/api/pith-number/JH6WLZO343ISMPDAIVEMG6MSPE/events.json","paper":"https://pith.science/paper/JH6WLZO3"},"agent_actions":{"view_html":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE","download_json":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE.json","view_paper":"https://pith.science/paper/JH6WLZO3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18010&json=true","fetch_graph":"https://pith.science/api/pith-number/JH6WLZO343ISMPDAIVEMG6MSPE/graph.json","fetch_events":"https://pith.science/api/pith-number/JH6WLZO343ISMPDAIVEMG6MSPE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE/action/storage_attestation","attest_author":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE/action/author_attestation","sign_citation":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE/action/citation_signature","submit_replication":"https://pith.science/pith/JH6WLZO343ISMPDAIVEMG6MSPE/action/replication_record"}},"created_at":"2026-07-05T07:47:25.253215+00:00","updated_at":"2026-07-05T07:47:25.253215+00:00"}