{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OM5DXVYBJQGEPN4ZQBLTRHJ4KL","short_pith_number":"pith:OM5DXVYB","schema_version":"1.0","canonical_sha256":"733a3bd7014c0c47b7998057389d3c52ef838b5ca8c2782213833e36929096a7","source":{"kind":"arxiv","id":"2307.10350","version":2},"attestation_state":"computed","paper":{"title":"Improving Multimodal Datasets with Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Gabriel Ilharco, Ludwig Schmidt, Samir Yitzhak Gadre, Sewoong Oh, Thao Nguyen","submitted_at":"2023-07-19T17:47:12Z","abstract_excerpt":"Massive web datasets play a key role in the success of large vision-language models like CLIP and Flamingo. However, the raw web data is noisy, and existing filtering methods to reduce noise often come at the expense of data diversity. Our work focuses on caption quality as one major source of noise, and studies how generated captions can increase the utility of web-scraped datapoints with nondescript text. Through exploring different mixing strategies for raw and generated captions, we outperform the best filtering method proposed by the DataComp benchmark by 2% on ImageNet and 4% on average "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.10350","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-07-19T17:47:12Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"969795508cf540ed72dd29493c0fb6e26c0caf819b083a4a6b5e09fb3d937c8e","abstract_canon_sha256":"835eb10f292b779f8ad0d41e5fa8ff3740be4a42e8bc473acfed17ed2ed7f1bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:12.697010Z","signature_b64":"yV/DNcnNGTZCJGhJHVdfxeksDQR8lJw87+nIgmLSW/sbcjL1HFKIjkOdpsQPt8s6fuzOyjNeaDJIMZWFr4ZgCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"733a3bd7014c0c47b7998057389d3c52ef838b5ca8c2782213833e36929096a7","last_reissued_at":"2026-07-05T07:05:12.696454Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:12.696454Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Multimodal Datasets with Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Gabriel Ilharco, Ludwig Schmidt, Samir Yitzhak Gadre, Sewoong Oh, Thao Nguyen","submitted_at":"2023-07-19T17:47:12Z","abstract_excerpt":"Massive web datasets play a key role in the success of large vision-language models like CLIP and Flamingo. However, the raw web data is noisy, and existing filtering methods to reduce noise often come at the expense of data diversity. Our work focuses on caption quality as one major source of noise, and studies how generated captions can increase the utility of web-scraped datapoints with nondescript text. Through exploring different mixing strategies for raw and generated captions, we outperform the best filtering method proposed by the DataComp benchmark by 2% on ImageNet and 4% on average "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.10350","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.10350/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.10350","created_at":"2026-07-05T07:05:12.696518+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.10350v2","created_at":"2026-07-05T07:05:12.696518+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.10350","created_at":"2026-07-05T07:05:12.696518+00:00"},{"alias_kind":"pith_short_12","alias_value":"OM5DXVYBJQGE","created_at":"2026-07-05T07:05:12.696518+00:00"},{"alias_kind":"pith_short_16","alias_value":"OM5DXVYBJQGEPN4Z","created_at":"2026-07-05T07:05:12.696518+00:00"},{"alias_kind":"pith_short_8","alias_value":"OM5DXVYB","created_at":"2026-07-05T07:05:12.696518+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2311.12793","citing_title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL","json":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL.json","graph_json":"https://pith.science/api/pith-number/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/graph.json","events_json":"https://pith.science/api/pith-number/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/events.json","paper":"https://pith.science/paper/OM5DXVYB"},"agent_actions":{"view_html":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL","download_json":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL.json","view_paper":"https://pith.science/paper/OM5DXVYB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.10350&json=true","fetch_graph":"https://pith.science/api/pith-number/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/graph.json","fetch_events":"https://pith.science/api/pith-number/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/action/storage_attestation","attest_author":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/action/author_attestation","sign_citation":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/action/citation_signature","submit_replication":"https://pith.science/pith/OM5DXVYBJQGEPN4ZQBLTRHJ4KL/action/replication_record"}},"created_at":"2026-07-05T07:05:12.696518+00:00","updated_at":"2026-07-05T07:05:12.696518+00:00"}