{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:A6MDE5KYBJRJ7PZ4HDAOAXLOOE","short_pith_number":"pith:A6MDE5KY","schema_version":"1.0","canonical_sha256":"07983275580a629fbf3c38c0e05d6e711d43fe05e23973a4404507a83fc21c28","source":{"kind":"arxiv","id":"2310.03211","version":2},"attestation_state":"computed","paper":{"title":"On the Performance of Multimodal Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Erhan Bas, Utsav Garg","submitted_at":"2023-10-04T23:33:36Z","abstract_excerpt":"Instruction-tuned large language models (LLMs) have demonstrated promising zero-shot generalization capabilities across various downstream tasks. Recent research has introduced multimodal capabilities to LLMs by integrating independently pretrained vision encoders through model grafting. These multimodal variants undergo instruction tuning, similar to LLMs, enabling effective zero-shot generalization for multimodal tasks. This study conducts a comparative analysis of different multimodal instruction tuning approaches and evaluates their performance across a range of tasks, including complex re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03211","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-04T23:33:36Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"f7ac6f38420c22fbd3dfe5889dea6a085b8d9f9392d6a294093069f4ca69ed33","abstract_canon_sha256":"70f826c64af24afc5d911a174116a27596560e12bbf1f9ca3416c5024165d02f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:32.401121Z","signature_b64":"IBmTrOH+WVowE/JVxa0lRuNf35Jptr9s6XO8MIdVMqQ4pxTwI/qzttZ8bFPXY2lNdL9hTemRaNUeJyRKSw1jCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07983275580a629fbf3c38c0e05d6e711d43fe05e23973a4404507a83fc21c28","last_reissued_at":"2026-07-05T07:17:32.400753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:32.400753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Performance of Multimodal Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Erhan Bas, Utsav Garg","submitted_at":"2023-10-04T23:33:36Z","abstract_excerpt":"Instruction-tuned large language models (LLMs) have demonstrated promising zero-shot generalization capabilities across various downstream tasks. Recent research has introduced multimodal capabilities to LLMs by integrating independently pretrained vision encoders through model grafting. These multimodal variants undergo instruction tuning, similar to LLMs, enabling effective zero-shot generalization for multimodal tasks. This study conducts a comparative analysis of different multimodal instruction tuning approaches and evaluates their performance across a range of tasks, including complex re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03211","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03211","created_at":"2026-07-05T07:17:32.400800+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03211v2","created_at":"2026-07-05T07:17:32.400800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03211","created_at":"2026-07-05T07:17:32.400800+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6MDE5KYBJRJ","created_at":"2026-07-05T07:17:32.400800+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6MDE5KYBJRJ7PZ4","created_at":"2026-07-05T07:17:32.400800+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6MDE5KY","created_at":"2026-07-05T07:17:32.400800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE","json":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE.json","graph_json":"https://pith.science/api/pith-number/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/graph.json","events_json":"https://pith.science/api/pith-number/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/events.json","paper":"https://pith.science/paper/A6MDE5KY"},"agent_actions":{"view_html":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE","download_json":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE.json","view_paper":"https://pith.science/paper/A6MDE5KY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03211&json=true","fetch_graph":"https://pith.science/api/pith-number/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/graph.json","fetch_events":"https://pith.science/api/pith-number/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/action/storage_attestation","attest_author":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/action/author_attestation","sign_citation":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/action/citation_signature","submit_replication":"https://pith.science/pith/A6MDE5KYBJRJ7PZ4HDAOAXLOOE/action/replication_record"}},"created_at":"2026-07-05T07:17:32.400800+00:00","updated_at":"2026-07-05T07:17:32.400800+00:00"}