{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GVMIA2JHN2VOKD3CV7IRN3AQHW","short_pith_number":"pith:GVMIA2JH","schema_version":"1.0","canonical_sha256":"35588069276eaae50f62afd116ec103da820cc85664172cd2c8c5f9e624527f2","source":{"kind":"arxiv","id":"2409.09086","version":1},"attestation_state":"computed","paper":{"title":"Inf-MLLM: Efficient Streaming Inference of Multimodal Large Language Models on a Single GPU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.DC","cs.PF"],"primary_cat":"cs.LG","authors_text":"Jieru Zhao, Minyi Guo, Qihao Jin, Wenchao Ding, Zhenyu Ning","submitted_at":"2024-09-11T12:44:12Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are distinguished by their multimodal comprehensive ability and widely used in many real-world applications including GPT-4o, autonomous driving and robotics. Despite their impressive performance, the multimodal inputs always incur long context. The inference under long context requires caching massive Key and Value states (KV cache) of previous tokens, which introduces high latency and excessive memory consumption. Due to this reason, it is challenging to deploy streaming inference of MLLMs on edge devices, which largely constrains the power and usage "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.09086","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-11T12:44:12Z","cross_cats_sorted":["cs.AI","cs.CV","cs.DC","cs.PF"],"title_canon_sha256":"0348812eed2677ca0be5ab687c817adc27ea85a56b1207c7e575d53031f29d98","abstract_canon_sha256":"9b5160d8006bb63e5db282ecbe50f7a75d51d9d1373640eb980c9a9e915df132"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:04.735309Z","signature_b64":"YcU22mGiLUl7i1AvEmsE3rsPZ15Dh12Cb0xETRsBXoViQ8wTcNkh0r/OUqBWoEyjQ5sbP8clOFXtyP8tQSAKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35588069276eaae50f62afd116ec103da820cc85664172cd2c8c5f9e624527f2","last_reissued_at":"2026-07-05T09:07:04.734918Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:04.734918Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inf-MLLM: Efficient Streaming Inference of Multimodal Large Language Models on a Single GPU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.DC","cs.PF"],"primary_cat":"cs.LG","authors_text":"Jieru Zhao, Minyi Guo, Qihao Jin, Wenchao Ding, Zhenyu Ning","submitted_at":"2024-09-11T12:44:12Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are distinguished by their multimodal comprehensive ability and widely used in many real-world applications including GPT-4o, autonomous driving and robotics. Despite their impressive performance, the multimodal inputs always incur long context. The inference under long context requires caching massive Key and Value states (KV cache) of previous tokens, which introduces high latency and excessive memory consumption. Due to this reason, it is challenging to deploy streaming inference of MLLMs on edge devices, which largely constrains the power and usage "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.09086","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.09086/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.09086","created_at":"2026-07-05T09:07:04.734976+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.09086v1","created_at":"2026-07-05T09:07:04.734976+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.09086","created_at":"2026-07-05T09:07:04.734976+00:00"},{"alias_kind":"pith_short_12","alias_value":"GVMIA2JHN2VO","created_at":"2026-07-05T09:07:04.734976+00:00"},{"alias_kind":"pith_short_16","alias_value":"GVMIA2JHN2VOKD3C","created_at":"2026-07-05T09:07:04.734976+00:00"},{"alias_kind":"pith_short_8","alias_value":"GVMIA2JH","created_at":"2026-07-05T09:07:04.734976+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":246,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25621","citing_title":"StreamOV: Streaming Omni-Video Understanding via Evidence-Guided Memory and Response Triggering","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23244","citing_title":"Convex Optimization for Alignment and Preference Learning on a Single GPU","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11128","citing_title":"Technology solutions targeting the performance of gen-AI inference in resource constrained platforms","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW","json":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW.json","graph_json":"https://pith.science/api/pith-number/GVMIA2JHN2VOKD3CV7IRN3AQHW/graph.json","events_json":"https://pith.science/api/pith-number/GVMIA2JHN2VOKD3CV7IRN3AQHW/events.json","paper":"https://pith.science/paper/GVMIA2JH"},"agent_actions":{"view_html":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW","download_json":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW.json","view_paper":"https://pith.science/paper/GVMIA2JH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.09086&json=true","fetch_graph":"https://pith.science/api/pith-number/GVMIA2JHN2VOKD3CV7IRN3AQHW/graph.json","fetch_events":"https://pith.science/api/pith-number/GVMIA2JHN2VOKD3CV7IRN3AQHW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW/action/storage_attestation","attest_author":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW/action/author_attestation","sign_citation":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW/action/citation_signature","submit_replication":"https://pith.science/pith/GVMIA2JHN2VOKD3CV7IRN3AQHW/action/replication_record"}},"created_at":"2026-07-05T09:07:04.734976+00:00","updated_at":"2026-07-05T09:07:04.734976+00:00"}