{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ACZBDNQM2QPK76MAFQXFN4SFQO","short_pith_number":"pith:ACZBDNQM","schema_version":"1.0","canonical_sha256":"00b211b60cd41eaff9802c2e56f2458388b3cf357ff50e88296559ec07013d41","source":{"kind":"arxiv","id":"2406.08155","version":2},"attestation_state":"computed","paper":{"title":"QuantMoE-Bench: Examining Post-Training Quantization for Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Pingzhi Li, Tianlong Chen, Xiaolong Jin, Yu Cheng, Zhen Tan","submitted_at":"2024-06-12T12:44:48Z","abstract_excerpt":"Mixture-of-Experts (MoE) is a promising way to scale up the learning capacity of large language models. It increases the number of parameters while keeping FLOPs nearly constant during inference through sparse activation. Yet, it still suffers from significant memory overheads due to the vast parameter size, necessitating model compression techniques. Post-training quantization offers a powerful approach for model compression. Existing methods adopt a fixed quantization precision for the entire MoE model. This rigid setup can lead to suboptimal performance, without considering the inherent spa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08155","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-12T12:44:48Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"934c37ffb2949aee9b4481715df0b3bb6675ca2c16216db35c3ac1dd84387aa2","abstract_canon_sha256":"5b49e2e58673e95b74e43f7642152667c87329fe405964d90691b958be441140"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:25.101921Z","signature_b64":"+maPKmAoKjEY5GrqnUVfZPyfD1AI9n+eyl/0MnJCuih1pK1jI3M/T+7erRRCuTsMKA3S81BwW2OKmU++Y+v8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00b211b60cd41eaff9802c2e56f2458388b3cf357ff50e88296559ec07013d41","last_reissued_at":"2026-07-05T10:19:25.101435Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:25.101435Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"QuantMoE-Bench: Examining Post-Training Quantization for Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Pingzhi Li, Tianlong Chen, Xiaolong Jin, Yu Cheng, Zhen Tan","submitted_at":"2024-06-12T12:44:48Z","abstract_excerpt":"Mixture-of-Experts (MoE) is a promising way to scale up the learning capacity of large language models. It increases the number of parameters while keeping FLOPs nearly constant during inference through sparse activation. Yet, it still suffers from significant memory overheads due to the vast parameter size, necessitating model compression techniques. Post-training quantization offers a powerful approach for model compression. Existing methods adopt a fixed quantization precision for the entire MoE model. This rigid setup can lead to suboptimal performance, without considering the inherent spa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08155","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08155/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08155","created_at":"2026-07-05T10:19:25.101496+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08155v2","created_at":"2026-07-05T10:19:25.101496+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08155","created_at":"2026-07-05T10:19:25.101496+00:00"},{"alias_kind":"pith_short_12","alias_value":"ACZBDNQM2QPK","created_at":"2026-07-05T10:19:25.101496+00:00"},{"alias_kind":"pith_short_16","alias_value":"ACZBDNQM2QPK76MA","created_at":"2026-07-05T10:19:25.101496+00:00"},{"alias_kind":"pith_short_8","alias_value":"ACZBDNQM","created_at":"2026-07-05T10:19:25.101496+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01444","citing_title":"On the Utility and Factual Reliability of Pruned Mixture-of-Experts Models in the Biomedical Domain","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27866","citing_title":"FlexMoE: One-for-All Nested Intra-Expert Pruning for MoE Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05688","citing_title":"Value-and-Structure Alignment for Routing-Consistent Quantization of Mixture-of-Experts Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10052","citing_title":"Swarm Skills: A Portable, Self-Evolving Multi-Agent System Specification for Coordination Engineering","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO","json":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO.json","graph_json":"https://pith.science/api/pith-number/ACZBDNQM2QPK76MAFQXFN4SFQO/graph.json","events_json":"https://pith.science/api/pith-number/ACZBDNQM2QPK76MAFQXFN4SFQO/events.json","paper":"https://pith.science/paper/ACZBDNQM"},"agent_actions":{"view_html":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO","download_json":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO.json","view_paper":"https://pith.science/paper/ACZBDNQM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08155&json=true","fetch_graph":"https://pith.science/api/pith-number/ACZBDNQM2QPK76MAFQXFN4SFQO/graph.json","fetch_events":"https://pith.science/api/pith-number/ACZBDNQM2QPK76MAFQXFN4SFQO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO/action/storage_attestation","attest_author":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO/action/author_attestation","sign_citation":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO/action/citation_signature","submit_replication":"https://pith.science/pith/ACZBDNQM2QPK76MAFQXFN4SFQO/action/replication_record"}},"created_at":"2026-07-05T10:19:25.101496+00:00","updated_at":"2026-07-05T10:19:25.101496+00:00"}