{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PAR6N5MP5PQI5JVHSITOEXYE4U","short_pith_number":"pith:PAR6N5MP","schema_version":"1.0","canonical_sha256":"7823e6f58febe08ea6a79226e25f04e52f25812f40781ca64ab93983ff637abe","source":{"kind":"arxiv","id":"2409.06166","version":1},"attestation_state":"computed","paper":{"title":"Revisiting Prompt Pretraining of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiajun Liang, Lingfeng Yang, Shuo Chen, Xiang Li, Zhaowei Chen, Zhenyuan Chen","submitted_at":"2024-09-10T02:36:13Z","abstract_excerpt":"Prompt learning is an effective method to customize Vision-Language Models (VLMs) for various downstream tasks, involving tuning very few parameters of input prompt tokens. Recently, prompt pretraining in large-scale dataset (e.g., ImageNet-21K) has played a crucial role in prompt learning for universal visual discrimination. However, we revisit and observe that the limited learnable prompts could face underfitting risks given the extensive images during prompt pretraining, simultaneously leading to poor generalization. To address the above issues, in this paper, we propose a general framework"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.06166","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-10T02:36:13Z","cross_cats_sorted":[],"title_canon_sha256":"1206cbba17c098ccc64c476c54eb5f848a73e97e72836f4b330f8381086d02e1","abstract_canon_sha256":"f72cf0de9167d836c9b52a9f28e43ca05819ac2046c4488160816d09a96b91a7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:03.003183Z","signature_b64":"1qkMHQ+72NOVVt1h9U6gegmh33F4NMtSDfUDxK6bPcNGQ0KzXqEZTQ/Bo6ubtRkxjHvuW8dUOGvTO8iojpI/Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7823e6f58febe08ea6a79226e25f04e52f25812f40781ca64ab93983ff637abe","last_reissued_at":"2026-07-05T09:05:03.002728Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:03.002728Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Prompt Pretraining of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiajun Liang, Lingfeng Yang, Shuo Chen, Xiang Li, Zhaowei Chen, Zhenyuan Chen","submitted_at":"2024-09-10T02:36:13Z","abstract_excerpt":"Prompt learning is an effective method to customize Vision-Language Models (VLMs) for various downstream tasks, involving tuning very few parameters of input prompt tokens. Recently, prompt pretraining in large-scale dataset (e.g., ImageNet-21K) has played a crucial role in prompt learning for universal visual discrimination. However, we revisit and observe that the limited learnable prompts could face underfitting risks given the extensive images during prompt pretraining, simultaneously leading to poor generalization. To address the above issues, in this paper, we propose a general framework"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06166","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06166/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.06166","created_at":"2026-07-05T09:05:03.002784+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.06166v1","created_at":"2026-07-05T09:05:03.002784+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06166","created_at":"2026-07-05T09:05:03.002784+00:00"},{"alias_kind":"pith_short_12","alias_value":"PAR6N5MP5PQI","created_at":"2026-07-05T09:05:03.002784+00:00"},{"alias_kind":"pith_short_16","alias_value":"PAR6N5MP5PQI5JVH","created_at":"2026-07-05T09:05:03.002784+00:00"},{"alias_kind":"pith_short_8","alias_value":"PAR6N5MP","created_at":"2026-07-05T09:05:03.002784+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.04732","citing_title":"LumiGen: An LVLM-Enhanced Iterative Framework for Fine-Grained Text-to-Image Generation","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U","json":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U.json","graph_json":"https://pith.science/api/pith-number/PAR6N5MP5PQI5JVHSITOEXYE4U/graph.json","events_json":"https://pith.science/api/pith-number/PAR6N5MP5PQI5JVHSITOEXYE4U/events.json","paper":"https://pith.science/paper/PAR6N5MP"},"agent_actions":{"view_html":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U","download_json":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U.json","view_paper":"https://pith.science/paper/PAR6N5MP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.06166&json=true","fetch_graph":"https://pith.science/api/pith-number/PAR6N5MP5PQI5JVHSITOEXYE4U/graph.json","fetch_events":"https://pith.science/api/pith-number/PAR6N5MP5PQI5JVHSITOEXYE4U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U/action/storage_attestation","attest_author":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U/action/author_attestation","sign_citation":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U/action/citation_signature","submit_replication":"https://pith.science/pith/PAR6N5MP5PQI5JVHSITOEXYE4U/action/replication_record"}},"created_at":"2026-07-05T09:05:03.002784+00:00","updated_at":"2026-07-05T09:05:03.002784+00:00"}