{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CJZGWNA35G37IIGFZWBSLYJXIP","short_pith_number":"pith:CJZGWNA3","schema_version":"1.0","canonical_sha256":"12726b341be9b7f420c5cd8325e13743c63b4dca170c65abc795a5f22fa58f2a","source":{"kind":"arxiv","id":"2402.11530","version":3},"attestation_state":"computed","paper":{"title":"Efficient Multimodal Learning from Data-centric Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boya Wu, Bo Zhao, Jianhao Yuan, Muyang He, Tiejun Huang, Yexin Liu, Yueze Wang","submitted_at":"2024-02-18T10:09:10Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated notable capabilities in general visual understanding and reasoning tasks. However, their deployment is hindered by substantial computational costs in both training and inference, limiting accessibility to the broader research and user communities. A straightforward solution is to leverage smaller pre-trained vision and language models, which inevitably cause significant performance drops. In this paper, we demonstrate the possibility of training a smaller but better MLLM with high-quality training data. Specifically, we introduce Bunny"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11530","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-18T10:09:10Z","cross_cats_sorted":[],"title_canon_sha256":"b2d6ece05912f9f741fb9846eee6a5ae8d29a786546106e375c567dd92a9b0b9","abstract_canon_sha256":"cd13aa3548e2b75afa2331929534e7624354d20707c72c7dcc4650d26bb8684a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:53.268790Z","signature_b64":"tq0nhSrD2IfbZbNJ+5QBFsoqdKSsJyEM1fQJooMTuUSz8qI0ncqgKk93r6A6XP82Ny+G7PdHfhFsj1PLEgGyCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12726b341be9b7f420c5cd8325e13743c63b4dca170c65abc795a5f22fa58f2a","last_reissued_at":"2026-07-05T08:46:53.268249Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:53.268249Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Multimodal Learning from Data-centric Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boya Wu, Bo Zhao, Jianhao Yuan, Muyang He, Tiejun Huang, Yexin Liu, Yueze Wang","submitted_at":"2024-02-18T10:09:10Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have demonstrated notable capabilities in general visual understanding and reasoning tasks. However, their deployment is hindered by substantial computational costs in both training and inference, limiting accessibility to the broader research and user communities. A straightforward solution is to leverage smaller pre-trained vision and language models, which inevitably cause significant performance drops. In this paper, we demonstrate the possibility of training a smaller but better MLLM with high-quality training data. Specifically, we introduce Bunny"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11530","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11530/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11530","created_at":"2026-07-05T08:46:53.268306+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11530v3","created_at":"2026-07-05T08:46:53.268306+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11530","created_at":"2026-07-05T08:46:53.268306+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJZGWNA35G37","created_at":"2026-07-05T08:46:53.268306+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJZGWNA35G37IIGF","created_at":"2026-07-05T08:46:53.268306+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJZGWNA3","created_at":"2026-07-05T08:46:53.268306+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":278,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12294","citing_title":"Bridging the Modality Gap in Forensic Image Retrieval","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11853","citing_title":"Task-Aware Structured Memory for Dynamic Multi-modal In-Context Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25813","citing_title":"Extending Embodied Question Answering from Perception to Decision","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07026","citing_title":"Modality Gap-Driven Subspace Alignment Training Paradigm For Multimodal Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10641","citing_title":"LLaVA-CKD: Bottom-Up Cascaded Knowledge Distillation for Vision-Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24380","citing_title":"Structural Pruning of Large Vision Language Models: A Comprehensive Study on Pruning Dynamics, Recovery, and Data Efficiency","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07726","citing_title":"PaliGemma: A versatile 3B VLM for transfer","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07825","citing_title":"Anisotropic Modality Align","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05649","citing_title":"Analogical Reasoning as a Doctor: A Foundation Model for Gastrointestinal Endoscopy Diagnosis","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01800","citing_title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP","json":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP.json","graph_json":"https://pith.science/api/pith-number/CJZGWNA35G37IIGFZWBSLYJXIP/graph.json","events_json":"https://pith.science/api/pith-number/CJZGWNA35G37IIGFZWBSLYJXIP/events.json","paper":"https://pith.science/paper/CJZGWNA3"},"agent_actions":{"view_html":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP","download_json":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP.json","view_paper":"https://pith.science/paper/CJZGWNA3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11530&json=true","fetch_graph":"https://pith.science/api/pith-number/CJZGWNA35G37IIGFZWBSLYJXIP/graph.json","fetch_events":"https://pith.science/api/pith-number/CJZGWNA35G37IIGFZWBSLYJXIP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP/action/storage_attestation","attest_author":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP/action/author_attestation","sign_citation":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP/action/citation_signature","submit_replication":"https://pith.science/pith/CJZGWNA35G37IIGFZWBSLYJXIP/action/replication_record"}},"created_at":"2026-07-05T08:46:53.268306+00:00","updated_at":"2026-07-05T08:46:53.268306+00:00"}