{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UVXSPQW3CQHCT2OKFOM724KSJ3","short_pith_number":"pith:UVXSPQW3","schema_version":"1.0","canonical_sha256":"a56f27c2db140e29e9ca2b99fd71524eee53d316c32486efa1a10bb5bd39ae84","source":{"kind":"arxiv","id":"2310.20550","version":3},"attestation_state":"computed","paper":{"title":"CapsFusion: Rethinking Image-Text Data at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Jingjing Liu, Qiying Yu, Quan Sun, Xiaosong Zhang, Xinlong Wang, Yue Cao, Yufeng Cui","submitted_at":"2023-10-31T15:31:39Z","abstract_excerpt":"Large multimodal models demonstrate remarkable generalist ability to perform diverse multimodal tasks in a zero-shot manner. Large-scale web-based image-text pairs contribute fundamentally to this success, but suffer from excessive noise. Recent studies use alternative captions synthesized by captioning models and have achieved notable benchmark performance. However, our experiments reveal significant Scalability Deficiency and World Knowledge Loss issues in models trained with synthetic captions, which have been largely obscured by their initial benchmark success. Upon closer examination, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.20550","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-31T15:31:39Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"e1c895b3c337f73aa101e73411174f3b19bba4f5b86334f5e1cdbfc8f65a9e76","abstract_canon_sha256":"91c1553a7b111a32bc5871187522d49169178ed3f66629d3be8ac7cf8fb8c272"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:34.872821Z","signature_b64":"ZTjPUbiwG/yPu8HWsMrXTvY0kPBqYAgfWgv8rp2WoSDp6W5Soi1dEbp/opGy+h+fcHMbbObzCk3nzE03YsyXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a56f27c2db140e29e9ca2b99fd71524eee53d316c32486efa1a10bb5bd39ae84","last_reissued_at":"2026-07-05T08:04:34.872299Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:34.872299Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CapsFusion: Rethinking Image-Text Data at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Jingjing Liu, Qiying Yu, Quan Sun, Xiaosong Zhang, Xinlong Wang, Yue Cao, Yufeng Cui","submitted_at":"2023-10-31T15:31:39Z","abstract_excerpt":"Large multimodal models demonstrate remarkable generalist ability to perform diverse multimodal tasks in a zero-shot manner. Large-scale web-based image-text pairs contribute fundamentally to this success, but suffer from excessive noise. Recent studies use alternative captions synthesized by captioning models and have achieved notable benchmark performance. However, our experiments reveal significant Scalability Deficiency and World Knowledge Loss issues in models trained with synthetic captions, which have been largely obscured by their initial benchmark success. Upon closer examination, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.20550","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.20550/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.20550","created_at":"2026-07-05T08:04:34.872374+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.20550v3","created_at":"2026-07-05T08:04:34.872374+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.20550","created_at":"2026-07-05T08:04:34.872374+00:00"},{"alias_kind":"pith_short_12","alias_value":"UVXSPQW3CQHC","created_at":"2026-07-05T08:04:34.872374+00:00"},{"alias_kind":"pith_short_16","alias_value":"UVXSPQW3CQHCT2OK","created_at":"2026-07-05T08:04:34.872374+00:00"},{"alias_kind":"pith_short_8","alias_value":"UVXSPQW3","created_at":"2026-07-05T08:04:34.872374+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":250,"is_internal_anchor":false},{"citing_arxiv_id":"2405.08748","citing_title":"Hunyuan-DiT: A Powerful Multi-Resolution Diffusion Transformer with Fine-Grained Chinese Understanding","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14396","citing_title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2403.05525","citing_title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3","json":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3.json","graph_json":"https://pith.science/api/pith-number/UVXSPQW3CQHCT2OKFOM724KSJ3/graph.json","events_json":"https://pith.science/api/pith-number/UVXSPQW3CQHCT2OKFOM724KSJ3/events.json","paper":"https://pith.science/paper/UVXSPQW3"},"agent_actions":{"view_html":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3","download_json":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3.json","view_paper":"https://pith.science/paper/UVXSPQW3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.20550&json=true","fetch_graph":"https://pith.science/api/pith-number/UVXSPQW3CQHCT2OKFOM724KSJ3/graph.json","fetch_events":"https://pith.science/api/pith-number/UVXSPQW3CQHCT2OKFOM724KSJ3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3/action/storage_attestation","attest_author":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3/action/author_attestation","sign_citation":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3/action/citation_signature","submit_replication":"https://pith.science/pith/UVXSPQW3CQHCT2OKFOM724KSJ3/action/replication_record"}},"created_at":"2026-07-05T08:04:34.872374+00:00","updated_at":"2026-07-05T08:04:34.872374+00:00"}