{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IX5LYQNLMYXRGEVPIJCYLETUOS","short_pith_number":"pith:IX5LYQNL","schema_version":"1.0","canonical_sha256":"45fabc41ab662f1312af424585927474a65466d672894446e01842f9e7ba3294","source":{"kind":"arxiv","id":"2405.08911","version":1},"attestation_state":"computed","paper":{"title":"CLIP with Quality Captions: A Strong Pretraining for Vision Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Fartash Faghri, Hadi Pouransari, Oncel Tuzel, Pavan Kumar Anasosalu Vasu","submitted_at":"2024-05-14T19:06:24Z","abstract_excerpt":"CLIP models perform remarkably well on zero-shot classification and retrieval tasks. But recent studies have shown that learnt representations in CLIP are not well suited for dense prediction tasks like object detection, semantic segmentation or depth estimation. More recently, multi-stage training methods for CLIP models was introduced to mitigate the weak performance of CLIP on downstream tasks. In this work, we find that simply improving the quality of captions in image-text datasets improves the quality of CLIP's visual representations, resulting in significant improvement on downstream de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.08911","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-14T19:06:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d0a6dac63b3e82b8fcf77d242c8fb5a052496fc3a9e15a72eb68d1e995a85f23","abstract_canon_sha256":"ac612aafc3b49473ed92403d4cc71665ad3948a101e3a2d601a7029f09e5cdb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:19:10.758814Z","signature_b64":"Bb75D2SMU+xt6VwOKc0R5vu92b+4AKATMpEb25F2ajF6yPXfwRDz0NBbGEqQ3be8aWhehBE32d3VQAiqU6eHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45fabc41ab662f1312af424585927474a65466d672894446e01842f9e7ba3294","last_reissued_at":"2026-07-05T08:19:10.758367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:19:10.758367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIP with Quality Captions: A Strong Pretraining for Vision Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Fartash Faghri, Hadi Pouransari, Oncel Tuzel, Pavan Kumar Anasosalu Vasu","submitted_at":"2024-05-14T19:06:24Z","abstract_excerpt":"CLIP models perform remarkably well on zero-shot classification and retrieval tasks. But recent studies have shown that learnt representations in CLIP are not well suited for dense prediction tasks like object detection, semantic segmentation or depth estimation. More recently, multi-stage training methods for CLIP models was introduced to mitigate the weak performance of CLIP on downstream tasks. In this work, we find that simply improving the quality of captions in image-text datasets improves the quality of CLIP's visual representations, resulting in significant improvement on downstream de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.08911","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.08911/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.08911","created_at":"2026-07-05T08:19:10.758425+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.08911v1","created_at":"2026-07-05T08:19:10.758425+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.08911","created_at":"2026-07-05T08:19:10.758425+00:00"},{"alias_kind":"pith_short_12","alias_value":"IX5LYQNLMYXR","created_at":"2026-07-05T08:19:10.758425+00:00"},{"alias_kind":"pith_short_16","alias_value":"IX5LYQNLMYXRGEVP","created_at":"2026-07-05T08:19:10.758425+00:00"},{"alias_kind":"pith_short_8","alias_value":"IX5LYQNL","created_at":"2026-07-05T08:19:10.758425+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.10372","citing_title":"UniMed-CLIP: Towards a Unified Image-Text Pretraining Paradigm for Diverse Medical Imaging Modalities","ref_index":81,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS","json":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS.json","graph_json":"https://pith.science/api/pith-number/IX5LYQNLMYXRGEVPIJCYLETUOS/graph.json","events_json":"https://pith.science/api/pith-number/IX5LYQNLMYXRGEVPIJCYLETUOS/events.json","paper":"https://pith.science/paper/IX5LYQNL"},"agent_actions":{"view_html":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS","download_json":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS.json","view_paper":"https://pith.science/paper/IX5LYQNL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.08911&json=true","fetch_graph":"https://pith.science/api/pith-number/IX5LYQNLMYXRGEVPIJCYLETUOS/graph.json","fetch_events":"https://pith.science/api/pith-number/IX5LYQNLMYXRGEVPIJCYLETUOS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS/action/storage_attestation","attest_author":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS/action/author_attestation","sign_citation":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS/action/citation_signature","submit_replication":"https://pith.science/pith/IX5LYQNLMYXRGEVPIJCYLETUOS/action/replication_record"}},"created_at":"2026-07-05T08:19:10.758425+00:00","updated_at":"2026-07-05T08:19:10.758425+00:00"}