{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7JFTDSNO6CSI4L55YFRV7GWHSH","short_pith_number":"pith:7JFTDSNO","schema_version":"1.0","canonical_sha256":"fa4b31c9aef0a48e2fbdc1635f9ac791c3b8864305df1d26cf1789628f5ba4ed","source":{"kind":"arxiv","id":"2502.00425","version":2},"attestation_state":"computed","paper":{"title":"MQuant: Unleashing the Inference Potential of Multimodal Large Language Models via Full Static Quantization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Changyong Shu, Chen Xu, Dawei Yang, Jiangyong Yu, Shuo Wang, Shuoyu Li, Sifan Zhou, Xing Hu, Zhihang Yuan, Zukang Xu","submitted_at":"2025-02-01T13:08:02Z","abstract_excerpt":"Multimodal large language models (MLLMs) have garnered widespread attention due to their ability to understand multimodal input. However, their large parameter sizes and substantial computational demands severely hinder their practical deployment and application.While quantization is an effective way to reduce model size and inference latency, its application to MLLMs remains underexplored. In this paper, we propose MQuant, a post-training quantization (PTQ) framework designed to tackle the unique challenges of multimodal large language models (MLLMs). Conventional quantization often struggles"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00425","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-01T13:08:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a485bce3cb6e45b523ddd72a985bf79b18f48eabab203520e76a2ad4fe7f8b6f","abstract_canon_sha256":"079dba9e5ad3a43280a4c84bcddcfd51fdffc9aa1fbc4e8670d832a369ed107a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:21.477992Z","signature_b64":"T99/hca967jIVtQ+DWdPrdc/Fw36oQ76f/5bG9mIt5YallXG5B3LrgG5y9OeYWuGDtHkC9w8CHfRVAxclrxFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa4b31c9aef0a48e2fbdc1635f9ac791c3b8864305df1d26cf1789628f5ba4ed","last_reissued_at":"2026-07-05T11:51:21.477461Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:21.477461Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MQuant: Unleashing the Inference Potential of Multimodal Large Language Models via Full Static Quantization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Changyong Shu, Chen Xu, Dawei Yang, Jiangyong Yu, Shuo Wang, Shuoyu Li, Sifan Zhou, Xing Hu, Zhihang Yuan, Zukang Xu","submitted_at":"2025-02-01T13:08:02Z","abstract_excerpt":"Multimodal large language models (MLLMs) have garnered widespread attention due to their ability to understand multimodal input. However, their large parameter sizes and substantial computational demands severely hinder their practical deployment and application.While quantization is an effective way to reduce model size and inference latency, its application to MLLMs remains underexplored. In this paper, we propose MQuant, a post-training quantization (PTQ) framework designed to tackle the unique challenges of multimodal large language models (MLLMs). Conventional quantization often struggles"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00425","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00425","created_at":"2026-07-05T11:51:21.477535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00425v2","created_at":"2026-07-05T11:51:21.477535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00425","created_at":"2026-07-05T11:51:21.477535+00:00"},{"alias_kind":"pith_short_12","alias_value":"7JFTDSNO6CSI","created_at":"2026-07-05T11:51:21.477535+00:00"},{"alias_kind":"pith_short_16","alias_value":"7JFTDSNO6CSI4L55","created_at":"2026-07-05T11:51:21.477535+00:00"},{"alias_kind":"pith_short_8","alias_value":"7JFTDSNO","created_at":"2026-07-05T11:51:21.477535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06579","citing_title":"Towards Efficient Multi-LLM Inference: Characterization and Analysis of LLM Routing and Hierarchical Techniques","ref_index":88,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH","json":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH.json","graph_json":"https://pith.science/api/pith-number/7JFTDSNO6CSI4L55YFRV7GWHSH/graph.json","events_json":"https://pith.science/api/pith-number/7JFTDSNO6CSI4L55YFRV7GWHSH/events.json","paper":"https://pith.science/paper/7JFTDSNO"},"agent_actions":{"view_html":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH","download_json":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH.json","view_paper":"https://pith.science/paper/7JFTDSNO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00425&json=true","fetch_graph":"https://pith.science/api/pith-number/7JFTDSNO6CSI4L55YFRV7GWHSH/graph.json","fetch_events":"https://pith.science/api/pith-number/7JFTDSNO6CSI4L55YFRV7GWHSH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH/action/storage_attestation","attest_author":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH/action/author_attestation","sign_citation":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH/action/citation_signature","submit_replication":"https://pith.science/pith/7JFTDSNO6CSI4L55YFRV7GWHSH/action/replication_record"}},"created_at":"2026-07-05T11:51:21.477535+00:00","updated_at":"2026-07-05T11:51:21.477535+00:00"}