{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GCKEAN2NP2NNKUEEFO6FC4QUW4","short_pith_number":"pith:GCKEAN2N","schema_version":"1.0","canonical_sha256":"309440374d7e9ad550842bbc517214b7332e5e89a1b0b9a87ce4f05149875930","source":{"kind":"arxiv","id":"2409.11055","version":6},"attestation_state":"computed","paper":{"title":"Exploring the Trade-Offs: Quantization Methods, Task Difficulty, and Model Size in Large Language Models From Edge to Giant","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jemin Lee, Jihun Oh, Jinse Kwon, Sihyeong Park, Yongin Kwon","submitted_at":"2024-09-17T10:31:37Z","abstract_excerpt":"Quantization has gained attention as a promising solution for the cost-effective deployment of large and small language models. However, most prior work has been limited to perplexity or basic knowledge tasks and lacks a comprehensive evaluation of recent models like Llama-3.3. In this paper, we conduct a comprehensive evaluation of instruction-tuned models spanning 1B to 405B parameters, applying four quantization methods across 13 datasets. Our findings reveal that (1) quantized models generally surpass smaller FP16 baselines, yet they often struggle with instruction-following and hallucinat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.11055","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-17T10:31:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"124fd1dc14a997531cfa95ff529db64abbf9d9cad7a40b70c24d3aa9e42bfe24","abstract_canon_sha256":"2fd8ad311d90ef33d6d431708e4942c8504ab72e13f2da9133623bb974142659"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:17.562677Z","signature_b64":"U3QHgs6VXS3Ou80pAMrB4v4HQPbhyKSjeeqBnVA7BMcJQ1QctFgrxoLxkswUYPB/mn+Q7SQNQo7VgwZd2nLaAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"309440374d7e9ad550842bbc517214b7332e5e89a1b0b9a87ce4f05149875930","last_reissued_at":"2026-07-05T11:15:17.562140Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:17.562140Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the Trade-Offs: Quantization Methods, Task Difficulty, and Model Size in Large Language Models From Edge to Giant","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jemin Lee, Jihun Oh, Jinse Kwon, Sihyeong Park, Yongin Kwon","submitted_at":"2024-09-17T10:31:37Z","abstract_excerpt":"Quantization has gained attention as a promising solution for the cost-effective deployment of large and small language models. However, most prior work has been limited to perplexity or basic knowledge tasks and lacks a comprehensive evaluation of recent models like Llama-3.3. In this paper, we conduct a comprehensive evaluation of instruction-tuned models spanning 1B to 405B parameters, applying four quantization methods across 13 datasets. Our findings reveal that (1) quantized models generally surpass smaller FP16 baselines, yet they often struggle with instruction-following and hallucinat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.11055","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.11055/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.11055","created_at":"2026-07-05T11:15:17.562193+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.11055v6","created_at":"2026-07-05T11:15:17.562193+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.11055","created_at":"2026-07-05T11:15:17.562193+00:00"},{"alias_kind":"pith_short_12","alias_value":"GCKEAN2NP2NN","created_at":"2026-07-05T11:15:17.562193+00:00"},{"alias_kind":"pith_short_16","alias_value":"GCKEAN2NP2NNKUEE","created_at":"2026-07-05T11:15:17.562193+00:00"},{"alias_kind":"pith_short_8","alias_value":"GCKEAN2N","created_at":"2026-07-05T11:15:17.562193+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08029","citing_title":"Rethinking Small VLM Quantization: From Component-Wise Analysis to Hardware-Aware Edge Deployment","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26558","citing_title":"Cassandra: Enabling Reasoning LLMs at Edge via Self-Speculative Decoding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19645","citing_title":"K-Quantization and its Impact on Output Performance","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28239","citing_title":"A Switch-Centric In-Network Architecture for Accelerating LLM Inference in Shared-Memory Network","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13806","citing_title":"Robust Ultra Low-Bit Post-Training Quantization via Stable Diagonal Curvature Estimate","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4","json":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4.json","graph_json":"https://pith.science/api/pith-number/GCKEAN2NP2NNKUEEFO6FC4QUW4/graph.json","events_json":"https://pith.science/api/pith-number/GCKEAN2NP2NNKUEEFO6FC4QUW4/events.json","paper":"https://pith.science/paper/GCKEAN2N"},"agent_actions":{"view_html":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4","download_json":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4.json","view_paper":"https://pith.science/paper/GCKEAN2N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.11055&json=true","fetch_graph":"https://pith.science/api/pith-number/GCKEAN2NP2NNKUEEFO6FC4QUW4/graph.json","fetch_events":"https://pith.science/api/pith-number/GCKEAN2NP2NNKUEEFO6FC4QUW4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4/action/storage_attestation","attest_author":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4/action/author_attestation","sign_citation":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4/action/citation_signature","submit_replication":"https://pith.science/pith/GCKEAN2NP2NNKUEEFO6FC4QUW4/action/replication_record"}},"created_at":"2026-07-05T11:15:17.562193+00:00","updated_at":"2026-07-05T11:15:17.562193+00:00"}