{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RQMW3AHMULXJTZVJ5MJYAZUDRT","short_pith_number":"pith:RQMW3AHM","schema_version":"1.0","canonical_sha256":"8c196d80eca2ee99e6a9eb138066838cd414f7b3d4e9555c2dd323906258d072","source":{"kind":"arxiv","id":"2402.10517","version":4},"attestation_state":"computed","paper":{"title":"Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bonggeun Sim, Jae W. Lee, Jake Hyun, SangLyul Cho, Yeonhong Park","submitted_at":"2024-02-16T09:06:06Z","abstract_excerpt":"Recently, considerable efforts have been directed towards compressing Large Language Models (LLMs), which showcase groundbreaking capabilities across diverse applications but entail significant deployment costs due to their large sizes. Meanwhile, much less attention has been given to mitigating the costs associated with deploying multiple LLMs of varying sizes despite its practical significance. Thus, this paper introduces \\emph{any-precision LLM}, extending the concept of any-precision DNN to LLMs. Addressing challenges in any-precision LLM, we propose a lightweight method for any-precision "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10517","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-16T09:06:06Z","cross_cats_sorted":[],"title_canon_sha256":"3f0487b35567502646d7ba0a3302be13cf58bb14db9d7f2786222252e1c417c7","abstract_canon_sha256":"3855a8f1741eb67ff5830b1da7890cc2c2cb6464f2485866431f50a4ae21d055"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:00.109502Z","signature_b64":"cUK/la71RslJ47xY4OORPqP+5wj0sDmSn7IYUEuvHMuaSLU4JgL7A7mCX2QMMKysO4gRRtBtq18502O9LJHXAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c196d80eca2ee99e6a9eb138066838cd414f7b3d4e9555c2dd323906258d072","last_reissued_at":"2026-07-05T08:35:00.109077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:00.109077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Any-Precision LLM: Low-Cost Deployment of Multiple, Different-Sized LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bonggeun Sim, Jae W. Lee, Jake Hyun, SangLyul Cho, Yeonhong Park","submitted_at":"2024-02-16T09:06:06Z","abstract_excerpt":"Recently, considerable efforts have been directed towards compressing Large Language Models (LLMs), which showcase groundbreaking capabilities across diverse applications but entail significant deployment costs due to their large sizes. Meanwhile, much less attention has been given to mitigating the costs associated with deploying multiple LLMs of varying sizes despite its practical significance. Thus, this paper introduces \\emph{any-precision LLM}, extending the concept of any-precision DNN to LLMs. Addressing challenges in any-precision LLM, we propose a lightweight method for any-precision "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10517","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10517/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10517","created_at":"2026-07-05T08:35:00.109133+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10517v4","created_at":"2026-07-05T08:35:00.109133+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10517","created_at":"2026-07-05T08:35:00.109133+00:00"},{"alias_kind":"pith_short_12","alias_value":"RQMW3AHMULXJ","created_at":"2026-07-05T08:35:00.109133+00:00"},{"alias_kind":"pith_short_16","alias_value":"RQMW3AHMULXJTZVJ","created_at":"2026-07-05T08:35:00.109133+00:00"},{"alias_kind":"pith_short_8","alias_value":"RQMW3AHM","created_at":"2026-07-05T08:35:00.109133+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23419","citing_title":"GRINQH: Graded Input-based Quantization Hierarchy for Efficient LLM Generation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12876","citing_title":"Multi-Bitwidth Quantization for LLMs Using Additive Codebooks","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26558","citing_title":"Cassandra: Enabling Reasoning LLMs at Edge via Self-Speculative Decoding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06836","citing_title":"STQuant: Spatio-Temporal Adaptive Framework for Optimizer Quantization in Large Multimodal Model Training","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT","json":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT.json","graph_json":"https://pith.science/api/pith-number/RQMW3AHMULXJTZVJ5MJYAZUDRT/graph.json","events_json":"https://pith.science/api/pith-number/RQMW3AHMULXJTZVJ5MJYAZUDRT/events.json","paper":"https://pith.science/paper/RQMW3AHM"},"agent_actions":{"view_html":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT","download_json":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT.json","view_paper":"https://pith.science/paper/RQMW3AHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10517&json=true","fetch_graph":"https://pith.science/api/pith-number/RQMW3AHMULXJTZVJ5MJYAZUDRT/graph.json","fetch_events":"https://pith.science/api/pith-number/RQMW3AHMULXJTZVJ5MJYAZUDRT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT/action/storage_attestation","attest_author":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT/action/author_attestation","sign_citation":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT/action/citation_signature","submit_replication":"https://pith.science/pith/RQMW3AHMULXJTZVJ5MJYAZUDRT/action/replication_record"}},"created_at":"2026-07-05T08:35:00.109133+00:00","updated_at":"2026-07-05T08:35:00.109133+00:00"}