{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6H72XOY4TEHHBTLE7OBHVPJBG4","short_pith_number":"pith:6H72XOY4","schema_version":"1.0","canonical_sha256":"f1ffabbb1c990e70cd64fb827abd213724690cee11f35691cc5a6ebb85c0dba6","source":{"kind":"arxiv","id":"2407.19409","version":1},"attestation_state":"computed","paper":{"title":"LLAVADI: What Matters For Multimodal Large Language Models Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Haobo Yuan, Lu Qi, Ming-Hsuan Yang, Shilin Xu, Xiangtai Li, Yunhai Tong","submitted_at":"2024-07-28T06:10:47Z","abstract_excerpt":"The recent surge in Multimodal Large Language Models (MLLMs) has showcased their remarkable potential for achieving generalized intelligence by integrating visual understanding into Large Language Models.Nevertheless, the sheer model size of MLLMs leads to substantial memory and computational demands that hinder their widespread deployment. In this work, we do not propose a new efficient model structure or train small-scale MLLMs from scratch. Instead, we focus on what matters for training small-scale MLLMs through knowledge distillation, which is the first step from the multimodal distillatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19409","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-28T06:10:47Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"ba0e676de736634e1701a367c365d4a6a29decca9f5e8239819fd3ef3805f3b0","abstract_canon_sha256":"bff74c48f0ed245aa9c263c2ce5dbe4db37d4587d35b28c5b09e1f2836ce2df7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:39.634662Z","signature_b64":"K0ChTB5Rk+Bj/GBBeNJmTJroKnW9Ps10EfPMNScarpGWmc1c/2P9rc6PbdeeQSBv+kMBNvd3Ldi56tz8m7V/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1ffabbb1c990e70cd64fb827abd213724690cee11f35691cc5a6ebb85c0dba6","last_reissued_at":"2026-07-05T08:49:39.634165Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:39.634165Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLAVADI: What Matters For Multimodal Large Language Models Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Haobo Yuan, Lu Qi, Ming-Hsuan Yang, Shilin Xu, Xiangtai Li, Yunhai Tong","submitted_at":"2024-07-28T06:10:47Z","abstract_excerpt":"The recent surge in Multimodal Large Language Models (MLLMs) has showcased their remarkable potential for achieving generalized intelligence by integrating visual understanding into Large Language Models.Nevertheless, the sheer model size of MLLMs leads to substantial memory and computational demands that hinder their widespread deployment. In this work, we do not propose a new efficient model structure or train small-scale MLLMs from scratch. Instead, we focus on what matters for training small-scale MLLMs through knowledge distillation, which is the first step from the multimodal distillatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19409","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19409/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19409","created_at":"2026-07-05T08:49:39.634223+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19409v1","created_at":"2026-07-05T08:49:39.634223+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19409","created_at":"2026-07-05T08:49:39.634223+00:00"},{"alias_kind":"pith_short_12","alias_value":"6H72XOY4TEHH","created_at":"2026-07-05T08:49:39.634223+00:00"},{"alias_kind":"pith_short_16","alias_value":"6H72XOY4TEHHBTLE","created_at":"2026-07-05T08:49:39.634223+00:00"},{"alias_kind":"pith_short_8","alias_value":"6H72XOY4","created_at":"2026-07-05T08:49:39.634223+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.21420","citing_title":"ReGATE: Learning Faster and Better with Fewer Tokens in MLLMs","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10641","citing_title":"LLaVA-CKD: Bottom-Up Cascaded Knowledge Distillation for Vision-Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":215,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4","json":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4.json","graph_json":"https://pith.science/api/pith-number/6H72XOY4TEHHBTLE7OBHVPJBG4/graph.json","events_json":"https://pith.science/api/pith-number/6H72XOY4TEHHBTLE7OBHVPJBG4/events.json","paper":"https://pith.science/paper/6H72XOY4"},"agent_actions":{"view_html":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4","download_json":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4.json","view_paper":"https://pith.science/paper/6H72XOY4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19409&json=true","fetch_graph":"https://pith.science/api/pith-number/6H72XOY4TEHHBTLE7OBHVPJBG4/graph.json","fetch_events":"https://pith.science/api/pith-number/6H72XOY4TEHHBTLE7OBHVPJBG4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4/action/storage_attestation","attest_author":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4/action/author_attestation","sign_citation":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4/action/citation_signature","submit_replication":"https://pith.science/pith/6H72XOY4TEHHBTLE7OBHVPJBG4/action/replication_record"}},"created_at":"2026-07-05T08:49:39.634223+00:00","updated_at":"2026-07-05T08:49:39.634223+00:00"}