{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MEO32AUE4W35D3CEARSTFR6I3D","short_pith_number":"pith:MEO32AUE","schema_version":"1.0","canonical_sha256":"611dbd0284e5b7d1ec44046532c7c8d8f86fff2e6ab69b75376b4194c515ee42","source":{"kind":"arxiv","id":"2405.12107","version":2},"attestation_state":"computed","paper":{"title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiajun Ding, Jun Yu, Lihao Zheng, Mingyang Wang, Xuecheng Ouyang, Zhenbiao Gai, Zhenwei Shao, Zhou Yu","submitted_at":"2024-05-20T15:23:19Z","abstract_excerpt":"By harnessing the capabilities of large language models (LLMs), recent large multimodal models (LMMs) have shown remarkable versatility in open-world multimodal understanding. Nevertheless, they are usually parameter-heavy and computation-intensive, thus hindering their applicability in resource-constrained scenarios. To this end, several lightweight LMMs have been proposed successively to maximize the capabilities under constrained scale (e.g., 3B). Despite the encouraging results achieved by these methods, most of them only focus on one or two aspects of the design space, and the key design "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12107","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-20T15:23:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8eb53ccfc52f074ea8bf4f37e481e5f78069fd8a163ead4fef3003d3aff1350d","abstract_canon_sha256":"a29c3903af0534d86ce37c94033f314a25e35a70b048a394415e8678a1f51e64"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:04.108691Z","signature_b64":"WbW4sh5yUU+XfcOa8xA7a67/QwklJNNksOodNU+ZRQWT3dKh+tLJcPMzhrteNZp3blp6Coc0igfnS7/v7MOWDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"611dbd0284e5b7d1ec44046532c7c8d8f86fff2e6ab69b75376b4194c515ee42","last_reissued_at":"2026-07-05T08:25:04.108190Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:04.108190Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiajun Ding, Jun Yu, Lihao Zheng, Mingyang Wang, Xuecheng Ouyang, Zhenbiao Gai, Zhenwei Shao, Zhou Yu","submitted_at":"2024-05-20T15:23:19Z","abstract_excerpt":"By harnessing the capabilities of large language models (LLMs), recent large multimodal models (LMMs) have shown remarkable versatility in open-world multimodal understanding. Nevertheless, they are usually parameter-heavy and computation-intensive, thus hindering their applicability in resource-constrained scenarios. To this end, several lightweight LMMs have been proposed successively to maximize the capabilities under constrained scale (e.g., 3B). Despite the encouraging results achieved by these methods, most of them only focus on one or two aspects of the design space, and the key design "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12107","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12107","created_at":"2026-07-05T08:25:04.108250+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12107v2","created_at":"2026-07-05T08:25:04.108250+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12107","created_at":"2026-07-05T08:25:04.108250+00:00"},{"alias_kind":"pith_short_12","alias_value":"MEO32AUE4W35","created_at":"2026-07-05T08:25:04.108250+00:00"},{"alias_kind":"pith_short_16","alias_value":"MEO32AUE4W35D3CE","created_at":"2026-07-05T08:25:04.108250+00:00"},{"alias_kind":"pith_short_8","alias_value":"MEO32AUE","created_at":"2026-07-05T08:25:04.108250+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10641","citing_title":"LLaVA-CKD: Bottom-Up Cascaded Knowledge Distillation for Vision-Language Models","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D","json":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D.json","graph_json":"https://pith.science/api/pith-number/MEO32AUE4W35D3CEARSTFR6I3D/graph.json","events_json":"https://pith.science/api/pith-number/MEO32AUE4W35D3CEARSTFR6I3D/events.json","paper":"https://pith.science/paper/MEO32AUE"},"agent_actions":{"view_html":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D","download_json":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D.json","view_paper":"https://pith.science/paper/MEO32AUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12107&json=true","fetch_graph":"https://pith.science/api/pith-number/MEO32AUE4W35D3CEARSTFR6I3D/graph.json","fetch_events":"https://pith.science/api/pith-number/MEO32AUE4W35D3CEARSTFR6I3D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D/action/storage_attestation","attest_author":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D/action/author_attestation","sign_citation":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D/action/citation_signature","submit_replication":"https://pith.science/pith/MEO32AUE4W35D3CEARSTFR6I3D/action/replication_record"}},"created_at":"2026-07-05T08:25:04.108250+00:00","updated_at":"2026-07-05T08:25:04.108250+00:00"}