{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WB7VTNONZJBT7C5EL7BCXIWM2B","short_pith_number":"pith:WB7VTNON","schema_version":"1.0","canonical_sha256":"b07f59b5cdca433f8ba45fc22ba2ccd06e33de9bb182183da440a39cfd2b7667","source":{"kind":"arxiv","id":"2307.02469","version":2},"attestation_state":"computed","paper":{"title":"What Matters in Training a GPT4-Style Language Model with Multimodal Inputs?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Guoqiang Wei, Hanbo Zhang, Jiangnan Xia, Jiani Zheng, Tao Kong, Yang Wei, Yan Zeng, Yuchen Zhang","submitted_at":"2023-07-05T17:44:28Z","abstract_excerpt":"Recent advancements in Large Language Models (LLMs) such as GPT4 have displayed exceptional multi-modal capabilities in following open-ended instructions given images. However, the performance of these models heavily relies on design choices such as network structures, training data, and training strategies, and these choices have not been extensively discussed in the literature, making it difficult to quantify progress in this field. To address this issue, this paper presents a systematic and comprehensive study, quantitatively and qualitatively, on training such models. We implement over 20 "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.02469","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-07-05T17:44:28Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"31cb6b5e0166cd3879c71afcf196d590fab969078bbd80d9e71151b3dc4cd3b0","abstract_canon_sha256":"1e752c4bd18a9c456b6f22ac861bc16c6dce87f05731608020482c1f45aa2276"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:52.647917Z","signature_b64":"+INbPcRV34LsZj6nvKUgWeSY1F4wbO2le9ysMldrhaytbL4std2RBVma0aVryweESFPKlVGTG0IX6DhciwEeDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b07f59b5cdca433f8ba45fc22ba2ccd06e33de9bb182183da440a39cfd2b7667","last_reissued_at":"2026-07-05T06:35:52.647434Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:52.647434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Matters in Training a GPT4-Style Language Model with Multimodal Inputs?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Guoqiang Wei, Hanbo Zhang, Jiangnan Xia, Jiani Zheng, Tao Kong, Yang Wei, Yan Zeng, Yuchen Zhang","submitted_at":"2023-07-05T17:44:28Z","abstract_excerpt":"Recent advancements in Large Language Models (LLMs) such as GPT4 have displayed exceptional multi-modal capabilities in following open-ended instructions given images. However, the performance of these models heavily relies on design choices such as network structures, training data, and training strategies, and these choices have not been extensively discussed in the literature, making it difficult to quantify progress in this field. To address this issue, this paper presents a systematic and comprehensive study, quantitatively and qualitatively, on training such models. We implement over 20 "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.02469","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.02469/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.02469","created_at":"2026-07-05T06:35:52.647494+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.02469v2","created_at":"2026-07-05T06:35:52.647494+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.02469","created_at":"2026-07-05T06:35:52.647494+00:00"},{"alias_kind":"pith_short_12","alias_value":"WB7VTNONZJBT","created_at":"2026-07-05T06:35:52.647494+00:00"},{"alias_kind":"pith_short_16","alias_value":"WB7VTNONZJBT7C5E","created_at":"2026-07-05T06:35:52.647494+00:00"},{"alias_kind":"pith_short_8","alias_value":"WB7VTNON","created_at":"2026-07-05T06:35:52.647494+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2309.15112","citing_title":"InternLM-XComposer: A Vision-Language Large Model for Advanced Text-image Comprehension and Composition","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2310.14566","citing_title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":173,"is_internal_anchor":false},{"citing_arxiv_id":"2402.12289","citing_title":"DriveVLM: The Convergence of Autonomous Driving and Large Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":203,"is_internal_anchor":false},{"citing_arxiv_id":"2308.06721","citing_title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13394","citing_title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B","json":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B.json","graph_json":"https://pith.science/api/pith-number/WB7VTNONZJBT7C5EL7BCXIWM2B/graph.json","events_json":"https://pith.science/api/pith-number/WB7VTNONZJBT7C5EL7BCXIWM2B/events.json","paper":"https://pith.science/paper/WB7VTNON"},"agent_actions":{"view_html":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B","download_json":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B.json","view_paper":"https://pith.science/paper/WB7VTNON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.02469&json=true","fetch_graph":"https://pith.science/api/pith-number/WB7VTNONZJBT7C5EL7BCXIWM2B/graph.json","fetch_events":"https://pith.science/api/pith-number/WB7VTNONZJBT7C5EL7BCXIWM2B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B/action/storage_attestation","attest_author":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B/action/author_attestation","sign_citation":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B/action/citation_signature","submit_replication":"https://pith.science/pith/WB7VTNONZJBT7C5EL7BCXIWM2B/action/replication_record"}},"created_at":"2026-07-05T06:35:52.647494+00:00","updated_at":"2026-07-05T06:35:52.647494+00:00"}