{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GOTNTHTUNIZDY3NJCIKKAXGWG4","short_pith_number":"pith:GOTNTHTU","schema_version":"1.0","canonical_sha256":"33a6d99e746a323c6da91214a05cd6370bba190f899a268cb97bf9bc277408e8","source":{"kind":"arxiv","id":"2306.15658","version":1},"attestation_state":"computed","paper":{"title":"CLIPA-v2: Scaling CLIP Training with 81.1% Zero-shot ImageNet Accuracy within a \\$10,000 Budget; An Extra \\$4,000 Unlocks 81.8% Accuracy","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cihang Xie, Xianhang Li, Zeyu Wang","submitted_at":"2023-06-27T17:51:06Z","abstract_excerpt":"The recent work CLIPA presents an inverse scaling law for CLIP training -- whereby the larger the image/text encoders used, the shorter the sequence length of image/text tokens that can be applied in training. This finding enables us to train high-performance CLIP models with significantly reduced computations. Building upon this work, we hereby present CLIPA-v2 with two key contributions. Technically, we find this inverse scaling law is also applicable in the finetuning stage, enabling further reduction in computational needs. Empirically, we explore CLIPA at scale, extending the experiments "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.15658","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2023-06-27T17:51:06Z","cross_cats_sorted":[],"title_canon_sha256":"9dab31eeec5f7984a6ce064f502d6111acf739fb411d7e14c6bf55cdd1390d10","abstract_canon_sha256":"70e0453a10667d4f9c3c0a2de8a92b23e818ee206733a88d0619bdc239d54f74"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:25:35.440635Z","signature_b64":"rPbbz9Z59HPDxGiQQMbuxlvJEBrN1giBlZul4TG1kKr8D+4Fzi8zjrjj6MYmHwjNZXa6KP34m7OCxD77AKI9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33a6d99e746a323c6da91214a05cd6370bba190f899a268cb97bf9bc277408e8","last_reissued_at":"2026-07-05T06:25:35.440134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:25:35.440134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIPA-v2: Scaling CLIP Training with 81.1% Zero-shot ImageNet Accuracy within a \\$10,000 Budget; An Extra \\$4,000 Unlocks 81.8% Accuracy","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cihang Xie, Xianhang Li, Zeyu Wang","submitted_at":"2023-06-27T17:51:06Z","abstract_excerpt":"The recent work CLIPA presents an inverse scaling law for CLIP training -- whereby the larger the image/text encoders used, the shorter the sequence length of image/text tokens that can be applied in training. This finding enables us to train high-performance CLIP models with significantly reduced computations. Building upon this work, we hereby present CLIPA-v2 with two key contributions. Technically, we find this inverse scaling law is also applicable in the finetuning stage, enabling further reduction in computational needs. Empirically, we explore CLIPA at scale, extending the experiments "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.15658","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.15658/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.15658","created_at":"2026-07-05T06:25:35.440197+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.15658v1","created_at":"2026-07-05T06:25:35.440197+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.15658","created_at":"2026-07-05T06:25:35.440197+00:00"},{"alias_kind":"pith_short_12","alias_value":"GOTNTHTUNIZD","created_at":"2026-07-05T06:25:35.440197+00:00"},{"alias_kind":"pith_short_16","alias_value":"GOTNTHTUNIZDY3NJ","created_at":"2026-07-05T06:25:35.440197+00:00"},{"alias_kind":"pith_short_8","alias_value":"GOTNTHTU","created_at":"2026-07-05T06:25:35.440197+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2303.15343","citing_title":"Sigmoid Loss for Language Image Pre-Training","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13181","citing_title":"Perception Encoder: The best visual embeddings are not at the output of the network","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2502.14786","citing_title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4","json":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4.json","graph_json":"https://pith.science/api/pith-number/GOTNTHTUNIZDY3NJCIKKAXGWG4/graph.json","events_json":"https://pith.science/api/pith-number/GOTNTHTUNIZDY3NJCIKKAXGWG4/events.json","paper":"https://pith.science/paper/GOTNTHTU"},"agent_actions":{"view_html":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4","download_json":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4.json","view_paper":"https://pith.science/paper/GOTNTHTU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.15658&json=true","fetch_graph":"https://pith.science/api/pith-number/GOTNTHTUNIZDY3NJCIKKAXGWG4/graph.json","fetch_events":"https://pith.science/api/pith-number/GOTNTHTUNIZDY3NJCIKKAXGWG4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4/action/storage_attestation","attest_author":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4/action/author_attestation","sign_citation":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4/action/citation_signature","submit_replication":"https://pith.science/pith/GOTNTHTUNIZDY3NJCIKKAXGWG4/action/replication_record"}},"created_at":"2026-07-05T06:25:35.440197+00:00","updated_at":"2026-07-05T06:25:35.440197+00:00"}