{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RP54NJOWATWZUWE57BSHU3XTSZ","short_pith_number":"pith:RP54NJOW","schema_version":"1.0","canonical_sha256":"8bfbc6a5d604ed9a589df8647a6ef3967a46c9a3703b5606d127d2a41cadf648","source":{"kind":"arxiv","id":"2402.09748","version":1},"attestation_state":"computed","paper":{"title":"Model Compression and Efficient Inference for Large Language Models: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"cs.CL","authors_text":"Binbin Lin, Deng Cai, Liye Zhang, Wei Chen, Wenxiao Wang, Xiaofei He, Yicong Luo, Yongliu Long, Zhengkai Lin","submitted_at":"2024-02-15T06:58:30Z","abstract_excerpt":"Transformer based large language models have achieved tremendous success. However, the significant memory and computational costs incurred during the inference process make it challenging to deploy large models on resource-constrained devices. In this paper, we investigate compression and efficient inference methods for large language models from an algorithmic perspective. Regarding taxonomy, similar to smaller models, compression and acceleration algorithms for large language models can still be categorized into quantization, pruning, distillation, compact architecture design, dynamic networ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.09748","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-15T06:58:30Z","cross_cats_sorted":["cs.AI","cs.LG","cs.PF"],"title_canon_sha256":"0e4dddf767962ece610bf48082e27dd6cd0242fd856e23648267f2211ae88bd2","abstract_canon_sha256":"dd89d289f0be661644f0a1733928a356cda9687f55aaca528196d917f2f641c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:45:36.362410Z","signature_b64":"17+bSqh+LE+oBNcR3maXTwrOfgj+N5ofse1SqqlWouphA3qpK9YvcBh3iaRAMWc/Z8sy0TFY8HtRbYzjAT49AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8bfbc6a5d604ed9a589df8647a6ef3967a46c9a3703b5606d127d2a41cadf648","last_reissued_at":"2026-07-05T07:45:36.361943Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:45:36.361943Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Compression and Efficient Inference for Large Language Models: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"cs.CL","authors_text":"Binbin Lin, Deng Cai, Liye Zhang, Wei Chen, Wenxiao Wang, Xiaofei He, Yicong Luo, Yongliu Long, Zhengkai Lin","submitted_at":"2024-02-15T06:58:30Z","abstract_excerpt":"Transformer based large language models have achieved tremendous success. However, the significant memory and computational costs incurred during the inference process make it challenging to deploy large models on resource-constrained devices. In this paper, we investigate compression and efficient inference methods for large language models from an algorithmic perspective. Regarding taxonomy, similar to smaller models, compression and acceleration algorithms for large language models can still be categorized into quantization, pruning, distillation, compact architecture design, dynamic networ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09748","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09748/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.09748","created_at":"2026-07-05T07:45:36.362003+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.09748v1","created_at":"2026-07-05T07:45:36.362003+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09748","created_at":"2026-07-05T07:45:36.362003+00:00"},{"alias_kind":"pith_short_12","alias_value":"RP54NJOWATWZ","created_at":"2026-07-05T07:45:36.362003+00:00"},{"alias_kind":"pith_short_16","alias_value":"RP54NJOWATWZUWE5","created_at":"2026-07-05T07:45:36.362003+00:00"},{"alias_kind":"pith_short_8","alias_value":"RP54NJOW","created_at":"2026-07-05T07:45:36.362003+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03328","citing_title":"Averaged Evaluation Masks Capability Trade-Offs: Multi-Source Calibration for High-Sparsity LLM Pruning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2508.03949","citing_title":"Model Compression vs. Adversarial Robustness: An Empirical Study on Language Models for Code","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02657","citing_title":"Less LLM, More Documents: Searching for Improved RAG","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ","json":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ.json","graph_json":"https://pith.science/api/pith-number/RP54NJOWATWZUWE57BSHU3XTSZ/graph.json","events_json":"https://pith.science/api/pith-number/RP54NJOWATWZUWE57BSHU3XTSZ/events.json","paper":"https://pith.science/paper/RP54NJOW"},"agent_actions":{"view_html":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ","download_json":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ.json","view_paper":"https://pith.science/paper/RP54NJOW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.09748&json=true","fetch_graph":"https://pith.science/api/pith-number/RP54NJOWATWZUWE57BSHU3XTSZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RP54NJOWATWZUWE57BSHU3XTSZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ/action/storage_attestation","attest_author":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ/action/author_attestation","sign_citation":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ/action/citation_signature","submit_replication":"https://pith.science/pith/RP54NJOWATWZUWE57BSHU3XTSZ/action/replication_record"}},"created_at":"2026-07-05T07:45:36.362003+00:00","updated_at":"2026-07-05T07:45:36.362003+00:00"}