{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:HA7S6BDKL2ZOSDROKN5H2ZAMBM","short_pith_number":"pith:HA7S6BDK","schema_version":"1.0","canonical_sha256":"383f2f046a5eb2e90e2e537a7d640c0b36964b374034ec3596fccfb1c8b3e3f8","source":{"kind":"arxiv","id":"2210.07183","version":2},"attestation_state":"computed","paper":{"title":"Visual Classification via Description from Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Carl Vondrick, Sachit Menon","submitted_at":"2022-10-13T17:03:46Z","abstract_excerpt":"Vision-language models (VLMs) such as CLIP have shown promising performance on a variety of recognition tasks using the standard zero-shot classification procedure -- computing similarity between the query image and the embedded words for each category. By only using the category name, they neglect to make use of the rich context of additional information that language affords. The procedure gives no intermediate understanding of why a category is chosen, and furthermore provides no mechanism for adjusting the criteria used towards this decision. We present an alternative framework for classif"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.07183","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-10-13T17:03:46Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1ef07264a0b159f12bf2794faf6ea4f3eee1b544a6cd36ceee851320db82617c","abstract_canon_sha256":"3a7bb7f215a2c0fc89908b4094bdb6055945c5f7ccabba5402047f4eb49de9b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:21:27.765481Z","signature_b64":"TfSIyBiCkn9WUGny+LRRoU89G+m+IImPzbho+oeO49Cg3W40qGul1TkRmXmEx1t3VtMqe+yVRTtL8tw390FeAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"383f2f046a5eb2e90e2e537a7d640c0b36964b374034ec3596fccfb1c8b3e3f8","last_reissued_at":"2026-07-05T05:21:27.765040Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:21:27.765040Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Classification via Description from Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Carl Vondrick, Sachit Menon","submitted_at":"2022-10-13T17:03:46Z","abstract_excerpt":"Vision-language models (VLMs) such as CLIP have shown promising performance on a variety of recognition tasks using the standard zero-shot classification procedure -- computing similarity between the query image and the embedded words for each category. By only using the category name, they neglect to make use of the rich context of additional information that language affords. The procedure gives no intermediate understanding of why a category is chosen, and furthermore provides no mechanism for adjusting the criteria used towards this decision. We present an alternative framework for classif"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.07183","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.07183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.07183","created_at":"2026-07-05T05:21:27.765089+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.07183v2","created_at":"2026-07-05T05:21:27.765089+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.07183","created_at":"2026-07-05T05:21:27.765089+00:00"},{"alias_kind":"pith_short_12","alias_value":"HA7S6BDKL2ZO","created_at":"2026-07-05T05:21:27.765089+00:00"},{"alias_kind":"pith_short_16","alias_value":"HA7S6BDKL2ZOSDRO","created_at":"2026-07-05T05:21:27.765089+00:00"},{"alias_kind":"pith_short_8","alias_value":"HA7S6BDK","created_at":"2026-07-05T05:21:27.765089+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26091","citing_title":"On-Policy Self-Distillation with Sampled Demonstrations Reduces Output Diversity","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23611","citing_title":"Data Selection Through Iterative Self-Filtering for Vision-Language Settings","ref_index":209,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21446","citing_title":"Synergistic Dual-Branch Adaptation for Multi-modal Generalized Category Discovery","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19584","citing_title":"Language-Instructed Vision Embeddings for Controllable and Generalizable Perception","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00684","citing_title":"AdaBoosting Text Prompts for Vision-Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08128","citing_title":"ViperGPT: Visual Inference via Python Execution for Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03390","citing_title":"Enhancing Self-Supervised Talking Head Forgery Detection via a Training-Free Dual-System Framework","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05344","citing_title":"Open-SAT: LLM-Guided Query Embedding Refinement for Open-Vocabulary Object Retrieval in Satellite Imagery","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21786","citing_title":"From Codebooks to VLMs: Evaluating Automated Visual Discourse Analysis for Climate Change on Social Media","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17629","citing_title":"BioVLM: Routing Prompts, Not Parameters, for Cross-Modality Generalization in Biomedical VLMs","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM","json":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM.json","graph_json":"https://pith.science/api/pith-number/HA7S6BDKL2ZOSDROKN5H2ZAMBM/graph.json","events_json":"https://pith.science/api/pith-number/HA7S6BDKL2ZOSDROKN5H2ZAMBM/events.json","paper":"https://pith.science/paper/HA7S6BDK"},"agent_actions":{"view_html":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM","download_json":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM.json","view_paper":"https://pith.science/paper/HA7S6BDK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.07183&json=true","fetch_graph":"https://pith.science/api/pith-number/HA7S6BDKL2ZOSDROKN5H2ZAMBM/graph.json","fetch_events":"https://pith.science/api/pith-number/HA7S6BDKL2ZOSDROKN5H2ZAMBM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM/action/storage_attestation","attest_author":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM/action/author_attestation","sign_citation":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM/action/citation_signature","submit_replication":"https://pith.science/pith/HA7S6BDKL2ZOSDROKN5H2ZAMBM/action/replication_record"}},"created_at":"2026-07-05T05:21:27.765089+00:00","updated_at":"2026-07-05T05:21:27.765089+00:00"}