{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OMHEN2T635FHDWPOY57YQ6SFOX","short_pith_number":"pith:OMHEN2T6","schema_version":"1.0","canonical_sha256":"730e46ea7edf4a71d9eec77f887a4575c69d7d92c54a660bf12a3e7589cb7a0d","source":{"kind":"arxiv","id":"2403.17007","version":1},"attestation_state":"computed","paper":{"title":"DreamLIP: Language-Image Pre-training with Long Captions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Lu, Kecheng Zheng, Shuailei Ma, Wei Chen, Wei Wu, Xin Jin, Yifei Zhang, Yujun Shen","submitted_at":"2024-03-25T17:59:42Z","abstract_excerpt":"Language-image pre-training largely relies on how precisely and thoroughly a text describes its paired image. In practice, however, the contents of an image can be so rich that well describing them requires lengthy captions (e.g., with 10 sentences), which are usually missing in existing datasets. Consequently, there are currently no clear evidences on whether and how language-image pre-training could benefit from long captions. To figure this out, we first re-caption 30M images with detailed descriptions using a pre-trained Multi-modality Large Language Model (MLLM), and then study the usage "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.17007","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-25T17:59:42Z","cross_cats_sorted":[],"title_canon_sha256":"77617be863844c6d0ca82ddf3a77da60b04441446c572cf29966aa64da198773","abstract_canon_sha256":"9ef5f64196f9bd6c46b54cbb9919d27a10eeb66133cefb12f5f7d2cf5b4e3895"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:23.841766Z","signature_b64":"ZT4H9rIVR31V4XsXpofHpx/k1P4uzAewjVKXBwW6zMtFY5UDlI/3u5ivKh5ewY98j9h3VpjJfzr1Z3HuXXQAAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"730e46ea7edf4a71d9eec77f887a4575c69d7d92c54a660bf12a3e7589cb7a0d","last_reissued_at":"2026-07-05T08:00:23.841271Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:23.841271Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DreamLIP: Language-Image Pre-training with Long Captions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Lu, Kecheng Zheng, Shuailei Ma, Wei Chen, Wei Wu, Xin Jin, Yifei Zhang, Yujun Shen","submitted_at":"2024-03-25T17:59:42Z","abstract_excerpt":"Language-image pre-training largely relies on how precisely and thoroughly a text describes its paired image. In practice, however, the contents of an image can be so rich that well describing them requires lengthy captions (e.g., with 10 sentences), which are usually missing in existing datasets. Consequently, there are currently no clear evidences on whether and how language-image pre-training could benefit from long captions. To figure this out, we first re-caption 30M images with detailed descriptions using a pre-trained Multi-modality Large Language Model (MLLM), and then study the usage "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.17007","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.17007/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.17007","created_at":"2026-07-05T08:00:23.841341+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.17007v1","created_at":"2026-07-05T08:00:23.841341+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.17007","created_at":"2026-07-05T08:00:23.841341+00:00"},{"alias_kind":"pith_short_12","alias_value":"OMHEN2T635FH","created_at":"2026-07-05T08:00:23.841341+00:00"},{"alias_kind":"pith_short_16","alias_value":"OMHEN2T635FHDWPO","created_at":"2026-07-05T08:00:23.841341+00:00"},{"alias_kind":"pith_short_8","alias_value":"OMHEN2T6","created_at":"2026-07-05T08:00:23.841341+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX","json":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX.json","graph_json":"https://pith.science/api/pith-number/OMHEN2T635FHDWPOY57YQ6SFOX/graph.json","events_json":"https://pith.science/api/pith-number/OMHEN2T635FHDWPOY57YQ6SFOX/events.json","paper":"https://pith.science/paper/OMHEN2T6"},"agent_actions":{"view_html":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX","download_json":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX.json","view_paper":"https://pith.science/paper/OMHEN2T6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.17007&json=true","fetch_graph":"https://pith.science/api/pith-number/OMHEN2T635FHDWPOY57YQ6SFOX/graph.json","fetch_events":"https://pith.science/api/pith-number/OMHEN2T635FHDWPOY57YQ6SFOX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX/action/storage_attestation","attest_author":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX/action/author_attestation","sign_citation":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX/action/citation_signature","submit_replication":"https://pith.science/pith/OMHEN2T635FHDWPOY57YQ6SFOX/action/replication_record"}},"created_at":"2026-07-05T08:00:23.841341+00:00","updated_at":"2026-07-05T08:00:23.841341+00:00"}