{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MUXVZ2YZPGB77M3O6OENS3H22Y","short_pith_number":"pith:MUXVZ2YZ","schema_version":"1.0","canonical_sha256":"652f5ceb197983ffb36ef388d96cfad61984cbe1046cd938d58c3281faa6db7e","source":{"kind":"arxiv","id":"2303.06865","version":2},"attestation_state":"computed","paper":{"title":"FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.PF"],"primary_cat":"cs.LG","authors_text":"Beidi Chen, Binhang Yuan, Ce Zhang, Christopher R\\'e, Clark Barrett, Daniel Y. Fu, Ion Stoica, Joseph E. Gonzalez, Lianmin Zheng, Max Ryabinin, Percy Liang, Ying Sheng, Zhiqiang Xie, Zhuohan Li","submitted_at":"2023-03-13T05:19:28Z","abstract_excerpt":"The high computational and memory requirements of large language model (LLM) inference make it feasible only with multiple high-end accelerators. Motivated by the emerging demand for latency-insensitive tasks with batched processing, this paper initiates the study of high-throughput LLM inference using limited resources, such as a single commodity GPU. We present FlexGen, a high-throughput generation engine for running LLMs with limited GPU memory. FlexGen can be flexibly configured under various hardware resource constraints by aggregating memory and computation from the GPU, CPU, and disk. B"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.06865","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-03-13T05:19:28Z","cross_cats_sorted":["cs.AI","cs.PF"],"title_canon_sha256":"d6cf220d577b076947108f6418a047e01538c181d91d0bba3ae45d3e1c4f5783","abstract_canon_sha256":"01af8b849fbe5aa21c23229c538b9a331283f0caaa4a0fcacce33f83e038dfb9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:19:39.855648Z","signature_b64":"7KziZ/NU3vgAV5PlLmaBhO7x7g1EX4TtQN+z8OPeUU1uzLxccygoBPI431+rukAgaPFAcg1AfyqDE1qHcK31AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"652f5ceb197983ffb36ef388d96cfad61984cbe1046cd938d58c3281faa6db7e","last_reissued_at":"2026-07-05T06:19:39.855124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:19:39.855124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.PF"],"primary_cat":"cs.LG","authors_text":"Beidi Chen, Binhang Yuan, Ce Zhang, Christopher R\\'e, Clark Barrett, Daniel Y. Fu, Ion Stoica, Joseph E. Gonzalez, Lianmin Zheng, Max Ryabinin, Percy Liang, Ying Sheng, Zhiqiang Xie, Zhuohan Li","submitted_at":"2023-03-13T05:19:28Z","abstract_excerpt":"The high computational and memory requirements of large language model (LLM) inference make it feasible only with multiple high-end accelerators. Motivated by the emerging demand for latency-insensitive tasks with batched processing, this paper initiates the study of high-throughput LLM inference using limited resources, such as a single commodity GPU. We present FlexGen, a high-throughput generation engine for running LLMs with limited GPU memory. FlexGen can be flexibly configured under various hardware resource constraints by aggregating memory and computation from the GPU, CPU, and disk. B"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.06865","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.06865/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.06865","created_at":"2026-07-05T06:19:39.855182+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.06865v2","created_at":"2026-07-05T06:19:39.855182+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.06865","created_at":"2026-07-05T06:19:39.855182+00:00"},{"alias_kind":"pith_short_12","alias_value":"MUXVZ2YZPGB7","created_at":"2026-07-05T06:19:39.855182+00:00"},{"alias_kind":"pith_short_16","alias_value":"MUXVZ2YZPGB77M3O","created_at":"2026-07-05T06:19:39.855182+00:00"},{"alias_kind":"pith_short_8","alias_value":"MUXVZ2YZ","created_at":"2026-07-05T06:19:39.855182+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":105,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24033","citing_title":"RoPE-Aware Bit Allocation for KV-Cache Quantization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01065","citing_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04032","citing_title":"Do Transformers Need Three Projections? Systematic Study of QKV Variants","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25698","citing_title":"Reference-Augmented Learning for Precise Tracking Policy of Tendon-Driven Continuum Robots","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24754","citing_title":"Motion-Compensated Weight Compression","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25550","citing_title":"DisagFusion: Asynchronous Pipeline Parallelism and Elastic Scheduling for Disaggregated Diffusion Serving","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28095","citing_title":"SiDP: Memory-Efficient Data Parallelism for Offline LLM Inference","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09508","citing_title":"From Rigid to Dynamic: Entropy-Guided Adaptive Inference for Long-Context LLMs","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2306.00978","citing_title":"AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20179","citing_title":"TIDE: Efficient and Lossless MoE Diffusion LLM Inference with I/O-aware Expert Offload","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11733","citing_title":"Position: LLM Inference Should Be Evaluated as Energy-to-Token Production","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2309.06180","citing_title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25699","citing_title":"NVLLM: A 3D NAND-Centric Architecture Enabling Edge on-Device LLM Inference","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07815","citing_title":"AsyncTLS: Efficient Generative LLM Inference with Asynchronous Two-level Sparse Attention","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y","json":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y.json","graph_json":"https://pith.science/api/pith-number/MUXVZ2YZPGB77M3O6OENS3H22Y/graph.json","events_json":"https://pith.science/api/pith-number/MUXVZ2YZPGB77M3O6OENS3H22Y/events.json","paper":"https://pith.science/paper/MUXVZ2YZ"},"agent_actions":{"view_html":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y","download_json":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y.json","view_paper":"https://pith.science/paper/MUXVZ2YZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.06865&json=true","fetch_graph":"https://pith.science/api/pith-number/MUXVZ2YZPGB77M3O6OENS3H22Y/graph.json","fetch_events":"https://pith.science/api/pith-number/MUXVZ2YZPGB77M3O6OENS3H22Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y/action/storage_attestation","attest_author":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y/action/author_attestation","sign_citation":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y/action/citation_signature","submit_replication":"https://pith.science/pith/MUXVZ2YZPGB77M3O6OENS3H22Y/action/replication_record"}},"created_at":"2026-07-05T06:19:39.855182+00:00","updated_at":"2026-07-05T06:19:39.855182+00:00"}