{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y34AZWCYFQTDIIISPENHJGCC6L","short_pith_number":"pith:Y34AZWCY","schema_version":"1.0","canonical_sha256":"c6f80cd8582c26342112791a749842f2d77d4cbf9902b42d11397a9ecad48518","source":{"kind":"arxiv","id":"2401.02330","version":4},"attestation_state":"computed","paper":{"title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jian Tang, Minjie Zhu, Ning Liu, Xiaofeng Mou, Yichen Zhu, Zhicai Ou","submitted_at":"2024-01-04T16:07:43Z","abstract_excerpt":"In this paper, we introduce LLaVA-$\\phi$ (LLaVA-Phi), an efficient multi-modal assistant that harnesses the power of the recently advanced small language model, Phi-2, to facilitate multi-modal dialogues. LLaVA-Phi marks a notable advancement in the realm of compact multi-modal models. It demonstrates that even smaller language models, with as few as 2.7B parameters, can effectively engage in intricate dialogues that integrate both textual and visual elements, provided they are trained with high-quality corpora. Our model delivers commendable performance on publicly available benchmarks that e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.02330","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-04T16:07:43Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"af58d89b74f7bb8cca5b6907909eba7e27b1b4525c472afd5eef29a181695a48","abstract_canon_sha256":"9af08bb35193edd4d1ada32e8574e4521c773038366dd574e2002ed857001302"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:02.329129Z","signature_b64":"tStESKtEPFpoQvgHcrmsgVwVNX8hPyay5bqkvWFjXcwkyrBKtLTThbJy6C18bcP2Bz4CrSk5B6Ki6e6bVotyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c6f80cd8582c26342112791a749842f2d77d4cbf9902b42d11397a9ecad48518","last_reissued_at":"2026-07-05T07:48:02.328664Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:02.328664Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jian Tang, Minjie Zhu, Ning Liu, Xiaofeng Mou, Yichen Zhu, Zhicai Ou","submitted_at":"2024-01-04T16:07:43Z","abstract_excerpt":"In this paper, we introduce LLaVA-$\\phi$ (LLaVA-Phi), an efficient multi-modal assistant that harnesses the power of the recently advanced small language model, Phi-2, to facilitate multi-modal dialogues. LLaVA-Phi marks a notable advancement in the realm of compact multi-modal models. It demonstrates that even smaller language models, with as few as 2.7B parameters, can effectively engage in intricate dialogues that integrate both textual and visual elements, provided they are trained with high-quality corpora. Our model delivers commendable performance on publicly available benchmarks that e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.02330","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02330/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.02330","created_at":"2026-07-05T07:48:02.328724+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.02330v4","created_at":"2026-07-05T07:48:02.328724+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02330","created_at":"2026-07-05T07:48:02.328724+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y34AZWCYFQTD","created_at":"2026-07-05T07:48:02.328724+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y34AZWCYFQTDIIIS","created_at":"2026-07-05T07:48:02.328724+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y34AZWCY","created_at":"2026-07-05T07:48:02.328724+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":285,"is_internal_anchor":false},{"citing_arxiv_id":"2402.11684","citing_title":"ALLaVA: Harnessing GPT4V-Synthesized Data for Lite Vision-Language Models","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03766","citing_title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12514","citing_title":"TinyVLA: Towards Fast, Data-Efficient Vision-Language-Action Models for Robotic Manipulation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05472","citing_title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14396","citing_title":"SEED-X: Multimodal Models with Unified Multi-granularity Comprehension and Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13848","citing_title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19790","citing_title":"From Plausibility to Verifiability: Risk-Controlled Generative OCR with Vision-Language Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2501.17811","citing_title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L","json":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L.json","graph_json":"https://pith.science/api/pith-number/Y34AZWCYFQTDIIISPENHJGCC6L/graph.json","events_json":"https://pith.science/api/pith-number/Y34AZWCYFQTDIIISPENHJGCC6L/events.json","paper":"https://pith.science/paper/Y34AZWCY"},"agent_actions":{"view_html":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L","download_json":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L.json","view_paper":"https://pith.science/paper/Y34AZWCY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.02330&json=true","fetch_graph":"https://pith.science/api/pith-number/Y34AZWCYFQTDIIISPENHJGCC6L/graph.json","fetch_events":"https://pith.science/api/pith-number/Y34AZWCYFQTDIIISPENHJGCC6L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L/action/storage_attestation","attest_author":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L/action/author_attestation","sign_citation":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L/action/citation_signature","submit_replication":"https://pith.science/pith/Y34AZWCYFQTDIIISPENHJGCC6L/action/replication_record"}},"created_at":"2026-07-05T07:48:02.328724+00:00","updated_at":"2026-07-05T07:48:02.328724+00:00"}