{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NNSMZUB4D37SXVF5CRQMDCVXIE","short_pith_number":"pith:NNSMZUB4","schema_version":"1.0","canonical_sha256":"6b64ccd03c1eff2bd4bd1460c18ab741183c1dbad3d5b8010e3d62bf009b7a19","source":{"kind":"arxiv","id":"2407.20454","version":2},"attestation_state":"computed","paper":{"title":"CoMMIT: Coordinated Multimodal Instruction Tuning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingbo Shang, Jiuxiang Gu, Julian McAuley, Junda Wu, Lina Yao, Tong Yu, Xiang Chen, Xintong Li, Yu Wang","submitted_at":"2024-07-29T23:18:55Z","abstract_excerpt":"Instruction tuning in multimodal large language models (MLLMs) generally involves cooperative learning between a backbone LLM and a feature encoder of non-text input modalities. The major challenge is how to efficiently find the synergy between the two modules so that LLMs can adapt their reasoning abilities to downstream tasks while feature encoders can adjust to provide more task-specific information about its modality. In this paper, we analyze the MLLM instruction tuning from both theoretical and empirical perspectives, where we find the unbalanced learning between the feature encoder and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20454","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-29T23:18:55Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"9fdf3f9e8ecb635708472dc0a69d8b6ff7955ac976ee21bbfd83e66ea3a92adf","abstract_canon_sha256":"6f8e515365078ec061b7e5a65f6a1dff700966b0f418e4540936d4150e932787"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:00.487501Z","signature_b64":"uPXggNaKaoWcU+xJH21xD+scH8mXuSHtZnB/HSoKFjhhmO+k8Tj9zKdrFkVWB4q0XsGjQ3RR2D7TARZ1maicBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b64ccd03c1eff2bd4bd1460c18ab741183c1dbad3d5b8010e3d62bf009b7a19","last_reissued_at":"2026-07-05T12:07:00.487013Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:00.487013Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoMMIT: Coordinated Multimodal Instruction Tuning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingbo Shang, Jiuxiang Gu, Julian McAuley, Junda Wu, Lina Yao, Tong Yu, Xiang Chen, Xintong Li, Yu Wang","submitted_at":"2024-07-29T23:18:55Z","abstract_excerpt":"Instruction tuning in multimodal large language models (MLLMs) generally involves cooperative learning between a backbone LLM and a feature encoder of non-text input modalities. The major challenge is how to efficiently find the synergy between the two modules so that LLMs can adapt their reasoning abilities to downstream tasks while feature encoders can adjust to provide more task-specific information about its modality. In this paper, we analyze the MLLM instruction tuning from both theoretical and empirical perspectives, where we find the unbalanced learning between the feature encoder and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20454","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20454/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20454","created_at":"2026-07-05T12:07:00.487071+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20454v2","created_at":"2026-07-05T12:07:00.487071+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20454","created_at":"2026-07-05T12:07:00.487071+00:00"},{"alias_kind":"pith_short_12","alias_value":"NNSMZUB4D37S","created_at":"2026-07-05T12:07:00.487071+00:00"},{"alias_kind":"pith_short_16","alias_value":"NNSMZUB4D37SXVF5","created_at":"2026-07-05T12:07:00.487071+00:00"},{"alias_kind":"pith_short_8","alias_value":"NNSMZUB4","created_at":"2026-07-05T12:07:00.487071+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17296","citing_title":"Pareto LoRA: Mitigating Modality Imbalance in Unified Multimodal Models via Pareto-Optimal Gradient Integration","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE","json":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE.json","graph_json":"https://pith.science/api/pith-number/NNSMZUB4D37SXVF5CRQMDCVXIE/graph.json","events_json":"https://pith.science/api/pith-number/NNSMZUB4D37SXVF5CRQMDCVXIE/events.json","paper":"https://pith.science/paper/NNSMZUB4"},"agent_actions":{"view_html":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE","download_json":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE.json","view_paper":"https://pith.science/paper/NNSMZUB4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20454&json=true","fetch_graph":"https://pith.science/api/pith-number/NNSMZUB4D37SXVF5CRQMDCVXIE/graph.json","fetch_events":"https://pith.science/api/pith-number/NNSMZUB4D37SXVF5CRQMDCVXIE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE/action/storage_attestation","attest_author":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE/action/author_attestation","sign_citation":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE/action/citation_signature","submit_replication":"https://pith.science/pith/NNSMZUB4D37SXVF5CRQMDCVXIE/action/replication_record"}},"created_at":"2026-07-05T12:07:00.487071+00:00","updated_at":"2026-07-05T12:07:00.487071+00:00"}