{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KXGEKNEU75MXSKDTJZ5CQYA3QZ","short_pith_number":"pith:KXGEKNEU","schema_version":"1.0","canonical_sha256":"55cc453494ff597928734e7a28601b864a942734883cc2eb9d31c084046adc92","source":{"kind":"arxiv","id":"2509.07727","version":1},"attestation_state":"computed","paper":{"title":"MoE-Compression: How the Compression Error of Experts Affects the Inference Accuracy of MoE Model?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Benben Liu, Dan Wang, Sheng Di, Songkai Ma, Xiaodong Yu, Xiaoyi Lu, Zhaorui Zhang","submitted_at":"2025-09-09T13:28:41Z","abstract_excerpt":"With the widespread application of Mixture of Experts (MoE) reasoning models in the field of LLM learning, efficiently serving MoE models under limited GPU memory constraints has emerged as a significant challenge. Offloading the non-activated experts to main memory has been identified as an efficient approach to address such a problem, while it brings the challenges of transferring the expert between the GPU memory and main memory. We need to explore an efficient approach to compress the expert and analyze how the compression error affects the inference performance.\n  To bridge this gap, we p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.07727","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-09T13:28:41Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"37205b55e84cac1dbf371937c75193ccf508d7e0bdabaed5d079e7565ff64e82","abstract_canon_sha256":"70be3011410d279479bb00ad42ae4a3edcb92d52818c922c2bb90e3accd1ae10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:27.125974Z","signature_b64":"FFdb1YESgFIMyWpOkJYHYftbqY/3rYVei9YK38AphKi97eQ7GXFN0Kij6owmx7RbFBDnzQpuVgT6ZS9WViT4Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"55cc453494ff597928734e7a28601b864a942734883cc2eb9d31c084046adc92","last_reissued_at":"2026-07-05T12:07:27.125440Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:27.125440Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoE-Compression: How the Compression Error of Experts Affects the Inference Accuracy of MoE Model?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Benben Liu, Dan Wang, Sheng Di, Songkai Ma, Xiaodong Yu, Xiaoyi Lu, Zhaorui Zhang","submitted_at":"2025-09-09T13:28:41Z","abstract_excerpt":"With the widespread application of Mixture of Experts (MoE) reasoning models in the field of LLM learning, efficiently serving MoE models under limited GPU memory constraints has emerged as a significant challenge. Offloading the non-activated experts to main memory has been identified as an efficient approach to address such a problem, while it brings the challenges of transferring the expert between the GPU memory and main memory. We need to explore an efficient approach to compress the expert and analyze how the compression error affects the inference performance.\n  To bridge this gap, we p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.07727","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.07727/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.07727","created_at":"2026-07-05T12:07:27.125500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.07727v1","created_at":"2026-07-05T12:07:27.125500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.07727","created_at":"2026-07-05T12:07:27.125500+00:00"},{"alias_kind":"pith_short_12","alias_value":"KXGEKNEU75MX","created_at":"2026-07-05T12:07:27.125500+00:00"},{"alias_kind":"pith_short_16","alias_value":"KXGEKNEU75MXSKDT","created_at":"2026-07-05T12:07:27.125500+00:00"},{"alias_kind":"pith_short_8","alias_value":"KXGEKNEU","created_at":"2026-07-05T12:07:27.125500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.26388","citing_title":"SplitFT: An Adaptive Federated Split Learning System For LLMs Fine-Tuning","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ","json":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ.json","graph_json":"https://pith.science/api/pith-number/KXGEKNEU75MXSKDTJZ5CQYA3QZ/graph.json","events_json":"https://pith.science/api/pith-number/KXGEKNEU75MXSKDTJZ5CQYA3QZ/events.json","paper":"https://pith.science/paper/KXGEKNEU"},"agent_actions":{"view_html":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ","download_json":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ.json","view_paper":"https://pith.science/paper/KXGEKNEU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.07727&json=true","fetch_graph":"https://pith.science/api/pith-number/KXGEKNEU75MXSKDTJZ5CQYA3QZ/graph.json","fetch_events":"https://pith.science/api/pith-number/KXGEKNEU75MXSKDTJZ5CQYA3QZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ/action/storage_attestation","attest_author":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ/action/author_attestation","sign_citation":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ/action/citation_signature","submit_replication":"https://pith.science/pith/KXGEKNEU75MXSKDTJZ5CQYA3QZ/action/replication_record"}},"created_at":"2026-07-05T12:07:27.125500+00:00","updated_at":"2026-07-05T12:07:27.125500+00:00"}