{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TPBSD4SKP5HLMK5KAHC3JGS5X6","short_pith_number":"pith:TPBSD4SK","schema_version":"1.0","canonical_sha256":"9bc321f24a7f4eb62baa01c5b49a5dbfac4d4b6f2a61187aa23bc49f7ea39e80","source":{"kind":"arxiv","id":"2412.18135","version":2},"attestation_state":"computed","paper":{"title":"LSAQ: Layer-Specific Adaptive Quantization for Large Language Model Deployment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bin Ji, Binrui Zeng, Jie Yu, Jun Ma, Shangwen Wang, Shasha Li, XiaoDong Liu, Xiaopeng Li, Xinran Hong, Yongtao Tang","submitted_at":"2024-12-24T03:43:15Z","abstract_excerpt":"As Large Language Models (LLMs) demonstrate exceptional performance across various domains, deploying LLMs on edge devices has emerged as a new trend. Quantization techniques, which reduce the size and memory requirements of LLMs, are effective for deploying LLMs on resource-limited edge devices. However, existing one-size-fits-all quantization methods often fail to dynamically adjust the memory requirements of LLMs, limiting their applications to practical edge devices with various computation resources. To tackle this issue, we propose Layer-Specific Adaptive Quantization (LSAQ), a system fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18135","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-24T03:43:15Z","cross_cats_sorted":[],"title_canon_sha256":"a1f80547d6ed51668720242cf96942ce65da9da603176b4e0e9e82c28f9d00f2","abstract_canon_sha256":"50710e66a7a924d1907c226a66892172a08da8d126b98b3ba29f13c558657766"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:04.675001Z","signature_b64":"w48Dt8HRVGUTrZGFyAeXPsidRLSuZJpgYZuHewXzDpf98BFlwEcvFRXqEl1JfLpTi+cFVOfqaKISogT5JVHiBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9bc321f24a7f4eb62baa01c5b49a5dbfac4d4b6f2a61187aa23bc49f7ea39e80","last_reissued_at":"2026-07-05T10:59:04.674514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:04.674514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LSAQ: Layer-Specific Adaptive Quantization for Large Language Model Deployment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bin Ji, Binrui Zeng, Jie Yu, Jun Ma, Shangwen Wang, Shasha Li, XiaoDong Liu, Xiaopeng Li, Xinran Hong, Yongtao Tang","submitted_at":"2024-12-24T03:43:15Z","abstract_excerpt":"As Large Language Models (LLMs) demonstrate exceptional performance across various domains, deploying LLMs on edge devices has emerged as a new trend. Quantization techniques, which reduce the size and memory requirements of LLMs, are effective for deploying LLMs on resource-limited edge devices. However, existing one-size-fits-all quantization methods often fail to dynamically adjust the memory requirements of LLMs, limiting their applications to practical edge devices with various computation resources. To tackle this issue, we propose Layer-Specific Adaptive Quantization (LSAQ), a system fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18135","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18135/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18135","created_at":"2026-07-05T10:59:04.674567+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18135v2","created_at":"2026-07-05T10:59:04.674567+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18135","created_at":"2026-07-05T10:59:04.674567+00:00"},{"alias_kind":"pith_short_12","alias_value":"TPBSD4SKP5HL","created_at":"2026-07-05T10:59:04.674567+00:00"},{"alias_kind":"pith_short_16","alias_value":"TPBSD4SKP5HLMK5K","created_at":"2026-07-05T10:59:04.674567+00:00"},{"alias_kind":"pith_short_8","alias_value":"TPBSD4SK","created_at":"2026-07-05T10:59:04.674567+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01378","citing_title":"CE-LoRA: Computation-Efficient LoRA Fine-Tuning for Language Models","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6","json":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6.json","graph_json":"https://pith.science/api/pith-number/TPBSD4SKP5HLMK5KAHC3JGS5X6/graph.json","events_json":"https://pith.science/api/pith-number/TPBSD4SKP5HLMK5KAHC3JGS5X6/events.json","paper":"https://pith.science/paper/TPBSD4SK"},"agent_actions":{"view_html":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6","download_json":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6.json","view_paper":"https://pith.science/paper/TPBSD4SK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18135&json=true","fetch_graph":"https://pith.science/api/pith-number/TPBSD4SKP5HLMK5KAHC3JGS5X6/graph.json","fetch_events":"https://pith.science/api/pith-number/TPBSD4SKP5HLMK5KAHC3JGS5X6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6/action/storage_attestation","attest_author":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6/action/author_attestation","sign_citation":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6/action/citation_signature","submit_replication":"https://pith.science/pith/TPBSD4SKP5HLMK5KAHC3JGS5X6/action/replication_record"}},"created_at":"2026-07-05T10:59:04.674567+00:00","updated_at":"2026-07-05T10:59:04.674567+00:00"}