{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DNQSLKYH7NXOEDCP3OWW2YRWY6","short_pith_number":"pith:DNQSLKYH","schema_version":"1.0","canonical_sha256":"1b6125ab07fb6ee20c4fdbad6d6236c7a5e2362056be3ce977f9cae2fb4c32c7","source":{"kind":"arxiv","id":"2401.02418","version":1},"attestation_state":"computed","paper":{"title":"Learning to Prompt with Text Only Supervision for Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Federico Tombari, Luc Van Gool, Muhammad Ferjad Naeem, Muhammad Uzair Khattak, Muzammal Naseer","submitted_at":"2024-01-04T18:59:49Z","abstract_excerpt":"Foundational vision-language models such as CLIP are becoming a new paradigm in vision, due to their excellent generalization abilities. However, adapting these models for downstream tasks while maintaining their generalization remains a challenge. In literature, one branch of methods adapts CLIP by learning prompts using visual information. While effective, most of these works require labeled data which is not practical, and often struggle to generalize towards new datasets due to over-fitting on the source data. An alternative approach resorts to training-free methods by generating class des"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.02418","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-04T18:59:49Z","cross_cats_sorted":[],"title_canon_sha256":"c938d85d468f2f856af7ff3a9303a4fcf528e6b24db626ba61ed36dca960d8e7","abstract_canon_sha256":"42dd315742a2df5e564619c531e5ad29422f4f131c27fc0cae7ecc7479edfbe8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:30:16.440562Z","signature_b64":"0aqzR3S4BBzwBJSEzowC65Ax340bqNsebl18RERYsGCOpMO/vTYxp8Uz2LSQ/SpZQIyfNGoTnFiUxDfxzuK9CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b6125ab07fb6ee20c4fdbad6d6236c7a5e2362056be3ce977f9cae2fb4c32c7","last_reissued_at":"2026-07-05T07:30:16.440156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:30:16.440156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Prompt with Text Only Supervision for Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Federico Tombari, Luc Van Gool, Muhammad Ferjad Naeem, Muhammad Uzair Khattak, Muzammal Naseer","submitted_at":"2024-01-04T18:59:49Z","abstract_excerpt":"Foundational vision-language models such as CLIP are becoming a new paradigm in vision, due to their excellent generalization abilities. However, adapting these models for downstream tasks while maintaining their generalization remains a challenge. In literature, one branch of methods adapts CLIP by learning prompts using visual information. While effective, most of these works require labeled data which is not practical, and often struggle to generalize towards new datasets due to over-fitting on the source data. An alternative approach resorts to training-free methods by generating class des"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.02418","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.02418","created_at":"2026-07-05T07:30:16.440205+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.02418v1","created_at":"2026-07-05T07:30:16.440205+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02418","created_at":"2026-07-05T07:30:16.440205+00:00"},{"alias_kind":"pith_short_12","alias_value":"DNQSLKYH7NXO","created_at":"2026-07-05T07:30:16.440205+00:00"},{"alias_kind":"pith_short_16","alias_value":"DNQSLKYH7NXOEDCP","created_at":"2026-07-05T07:30:16.440205+00:00"},{"alias_kind":"pith_short_8","alias_value":"DNQSLKYH","created_at":"2026-07-05T07:30:16.440205+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.06624","citing_title":"Chameleon: Benchmarking Detection and Backtracking on Commercial-Grade AI-Generated Videos","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02753","citing_title":"DeCo-DETR: Decoupled Cognition DETR for efficient Open-Vocabulary Object Detection","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02753","citing_title":"DeCo-DETR: Decoupled Cognition DETR for efficient Open-Vocabulary Object Detection","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08156","citing_title":"LAGO: Language-Guided Adaptive Object-Region Focus for Zero-Shot Visual-Text Alignment","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6","json":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6.json","graph_json":"https://pith.science/api/pith-number/DNQSLKYH7NXOEDCP3OWW2YRWY6/graph.json","events_json":"https://pith.science/api/pith-number/DNQSLKYH7NXOEDCP3OWW2YRWY6/events.json","paper":"https://pith.science/paper/DNQSLKYH"},"agent_actions":{"view_html":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6","download_json":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6.json","view_paper":"https://pith.science/paper/DNQSLKYH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.02418&json=true","fetch_graph":"https://pith.science/api/pith-number/DNQSLKYH7NXOEDCP3OWW2YRWY6/graph.json","fetch_events":"https://pith.science/api/pith-number/DNQSLKYH7NXOEDCP3OWW2YRWY6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6/action/storage_attestation","attest_author":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6/action/author_attestation","sign_citation":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6/action/citation_signature","submit_replication":"https://pith.science/pith/DNQSLKYH7NXOEDCP3OWW2YRWY6/action/replication_record"}},"created_at":"2026-07-05T07:30:16.440205+00:00","updated_at":"2026-07-05T07:30:16.440205+00:00"}