{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LDULQC7KGHWZCJAZLUAUYM4QMA","short_pith_number":"pith:LDULQC7K","schema_version":"1.0","canonical_sha256":"58e8b80bea31ed9124195d014c3390603f73b4455a6a1c418668f3dfa72bfb08","source":{"kind":"arxiv","id":"2505.03804","version":1},"attestation_state":"computed","paper":{"title":"MoEQuant: Enhancing Quantization for Mixture-of-Experts Large Language Models via Expert-Balanced Sampling and Affinity Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chen Xu, Dawei Yang, Jiangyong Yu, Sifan Zhou, Xing Hu, Zhihang Yuan, Zhixuan Chen, Zukang Xu","submitted_at":"2025-05-02T08:51:55Z","abstract_excerpt":"Mixture-of-Experts (MoE) large language models (LLMs), which leverage dynamic routing and sparse activation to enhance efficiency and scalability, have achieved higher performance while reducing computational costs. However, these models face significant memory overheads, limiting their practical deployment and broader adoption. Post-training quantization (PTQ), a widely used method for compressing LLMs, encounters severe accuracy degradation and diminished generalization performance when applied to MoE models. This paper investigates the impact of MoE's sparse and dynamic characteristics on q"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03804","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-02T08:51:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5d040b7b723fff59a8ed2ad8bcedb51263fc79093b6260716190bbabbdead14b","abstract_canon_sha256":"89a3e26ce26489e2ffa9c24111a6b525f38efbf009a5e612e93718d352d80df1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:31.845794Z","signature_b64":"ul886Wmp+NOixUuJprKQ4BI3sH9VobiJZFzwwF1cV3w2Hd3EKvS2j/nMtJ9IFbKhj9IvCnIyti/V+WzSadH3BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58e8b80bea31ed9124195d014c3390603f73b4455a6a1c418668f3dfa72bfb08","last_reissued_at":"2026-07-05T10:59:31.845219Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:31.845219Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoEQuant: Enhancing Quantization for Mixture-of-Experts Large Language Models via Expert-Balanced Sampling and Affinity Guidance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chen Xu, Dawei Yang, Jiangyong Yu, Sifan Zhou, Xing Hu, Zhihang Yuan, Zhixuan Chen, Zukang Xu","submitted_at":"2025-05-02T08:51:55Z","abstract_excerpt":"Mixture-of-Experts (MoE) large language models (LLMs), which leverage dynamic routing and sparse activation to enhance efficiency and scalability, have achieved higher performance while reducing computational costs. However, these models face significant memory overheads, limiting their practical deployment and broader adoption. Post-training quantization (PTQ), a widely used method for compressing LLMs, encounters severe accuracy degradation and diminished generalization performance when applied to MoE models. This paper investigates the impact of MoE's sparse and dynamic characteristics on q"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03804","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03804/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03804","created_at":"2026-07-05T10:59:31.845294+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03804v1","created_at":"2026-07-05T10:59:31.845294+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03804","created_at":"2026-07-05T10:59:31.845294+00:00"},{"alias_kind":"pith_short_12","alias_value":"LDULQC7KGHWZ","created_at":"2026-07-05T10:59:31.845294+00:00"},{"alias_kind":"pith_short_16","alias_value":"LDULQC7KGHWZCJAZ","created_at":"2026-07-05T10:59:31.845294+00:00"},{"alias_kind":"pith_short_8","alias_value":"LDULQC7K","created_at":"2026-07-05T10:59:31.845294+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04980","citing_title":"AlphaQ: Calibration-Free Bit Allocation for Mixture-of-Experts Quantization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05688","citing_title":"Value-and-Structure Alignment for Routing-Consistent Quantization of Mixture-of-Experts Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23078","citing_title":"GEMQ: Global Expert-Level Mixed-Precision Quantization for MoE LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19929","citing_title":"Breaking Modality Heterogeneity in Low-Bit Quantization for Large Vision-Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18556","citing_title":"GSQ: Highly-Accurate Low-Precision Scalar Quantization for LLMs via Gumbel-Softmax Sampling","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07317","citing_title":"Amortized-Precision Quantization for Early-Exit Vision Transformers","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18556","citing_title":"GSQ: Highly-Accurate Low-Precision Scalar Quantization for LLMs via Gumbel-Softmax Sampling","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA","json":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA.json","graph_json":"https://pith.science/api/pith-number/LDULQC7KGHWZCJAZLUAUYM4QMA/graph.json","events_json":"https://pith.science/api/pith-number/LDULQC7KGHWZCJAZLUAUYM4QMA/events.json","paper":"https://pith.science/paper/LDULQC7K"},"agent_actions":{"view_html":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA","download_json":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA.json","view_paper":"https://pith.science/paper/LDULQC7K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03804&json=true","fetch_graph":"https://pith.science/api/pith-number/LDULQC7KGHWZCJAZLUAUYM4QMA/graph.json","fetch_events":"https://pith.science/api/pith-number/LDULQC7KGHWZCJAZLUAUYM4QMA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA/action/storage_attestation","attest_author":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA/action/author_attestation","sign_citation":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA/action/citation_signature","submit_replication":"https://pith.science/pith/LDULQC7KGHWZCJAZLUAUYM4QMA/action/replication_record"}},"created_at":"2026-07-05T10:59:31.845294+00:00","updated_at":"2026-07-05T10:59:31.845294+00:00"}