{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Z72L7Q26CFMFALOEWHNQ7LATTZ","short_pith_number":"pith:Z72L7Q26","schema_version":"1.0","canonical_sha256":"cff4bfc35e1158502dc4b1db0fac139e72b84b78333fcd34c1b22a5d1c53fa5b","source":{"kind":"arxiv","id":"2309.16414","version":3},"attestation_state":"computed","paper":{"title":"AutoCLIP: Auto-tuning Zero-Shot Classifiers for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chaithanya Kumar Mummadi, Jan Hendrik Metzen, Piyapat Saranrittichai","submitted_at":"2023-09-28T13:08:08Z","abstract_excerpt":"Classifiers built upon vision-language models such as CLIP have shown remarkable zero-shot performance across a broad range of image classification tasks. Prior work has studied different ways of automatically creating descriptor sets for every class based on prompt templates, ranging from manually engineered templates over templates obtained from a large language model to templates built from random words and characters. Up until now, deriving zero-shot classifiers from the respective encoded class descriptors has remained nearly unchanged, i.e., classify to the class that maximizes cosine si"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16414","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-28T13:08:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"48400a67b9533e563bc31c36351359ca4f23a8155c2ec67deaaa87a87abac7c9","abstract_canon_sha256":"c6a978c20fe9d513216d9b9f53629c064dcdbd293c8a34a1b95509589d465e2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:20.705102Z","signature_b64":"1X0VYb1gGpTSbB6M3fgmPHjR4UyYnYBBTHh5oFXXFN3VfjRO23v/JIm9EPs9kCI0CA61vFCGe4RB9djUAnhfDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cff4bfc35e1158502dc4b1db0fac139e72b84b78333fcd34c1b22a5d1c53fa5b","last_reissued_at":"2026-07-05T08:55:20.704757Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:20.704757Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoCLIP: Auto-tuning Zero-Shot Classifiers for Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chaithanya Kumar Mummadi, Jan Hendrik Metzen, Piyapat Saranrittichai","submitted_at":"2023-09-28T13:08:08Z","abstract_excerpt":"Classifiers built upon vision-language models such as CLIP have shown remarkable zero-shot performance across a broad range of image classification tasks. Prior work has studied different ways of automatically creating descriptor sets for every class based on prompt templates, ranging from manually engineered templates over templates obtained from a large language model to templates built from random words and characters. Up until now, deriving zero-shot classifiers from the respective encoded class descriptors has remained nearly unchanged, i.e., classify to the class that maximizes cosine si"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16414","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16414","created_at":"2026-07-05T08:55:20.704812+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16414v3","created_at":"2026-07-05T08:55:20.704812+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16414","created_at":"2026-07-05T08:55:20.704812+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z72L7Q26CFMF","created_at":"2026-07-05T08:55:20.704812+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z72L7Q26CFMFALOE","created_at":"2026-07-05T08:55:20.704812+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z72L7Q26","created_at":"2026-07-05T08:55:20.704812+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.15576","citing_title":"Visual Perturbation and Adaptive Hard Negative Contrastive Learning for Compositional Reasoning in Vision-Language Models","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ","json":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ.json","graph_json":"https://pith.science/api/pith-number/Z72L7Q26CFMFALOEWHNQ7LATTZ/graph.json","events_json":"https://pith.science/api/pith-number/Z72L7Q26CFMFALOEWHNQ7LATTZ/events.json","paper":"https://pith.science/paper/Z72L7Q26"},"agent_actions":{"view_html":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ","download_json":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ.json","view_paper":"https://pith.science/paper/Z72L7Q26","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16414&json=true","fetch_graph":"https://pith.science/api/pith-number/Z72L7Q26CFMFALOEWHNQ7LATTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/Z72L7Q26CFMFALOEWHNQ7LATTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ/action/storage_attestation","attest_author":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ/action/author_attestation","sign_citation":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ/action/citation_signature","submit_replication":"https://pith.science/pith/Z72L7Q26CFMFALOEWHNQ7LATTZ/action/replication_record"}},"created_at":"2026-07-05T08:55:20.704812+00:00","updated_at":"2026-07-05T08:55:20.704812+00:00"}