{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7S2H6CM3WWJ45KMDZTQ5YUUORS","short_pith_number":"pith:7S2H6CM3","schema_version":"1.0","canonical_sha256":"fcb47f099bb593cea983cce1dc528e8c8dd5f262cf99a61c35e2aab8f46f6b94","source":{"kind":"arxiv","id":"2310.18859","version":2},"attestation_state":"computed","paper":{"title":"SiDA-MoE: Sparsity-Inspired Data-Aware Serving for Efficient and Scalable Large Mixture-of-Experts Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Ang Li, Hai \"Helen\" Li, Jingwei Sun, Qilin Zheng, Shiyu Li, Xiangyu Jiang, Yiran Chen, Yongkai Wu, Yuhao Wu, Zhixu Du","submitted_at":"2023-10-29T01:08:55Z","abstract_excerpt":"Mixture-of-Experts (MoE) has emerged as a favorable architecture in the era of large models due to its inherent advantage, i.e., enlarging model capacity without incurring notable computational overhead. Yet, the realization of such benefits often results in ineffective GPU memory utilization, as large portions of the model parameters remain dormant during inference. Moreover, the memory demands of large models consistently outpace the memory capacity of contemporary GPUs. Addressing this, we introduce SiDA-MoE ($\\textbf{S}$parsity-$\\textbf{i}$nspired $\\textbf{D}$ata-$\\textbf{A}$ware), an effi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.18859","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-29T01:08:55Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"df25dfac2b57f54924c8e2b9d401b37e5cdcfcf739c6bd56e046d480b284d40f","abstract_canon_sha256":"efb65594832201dae43e1b6e232dbc9438122cfd8f2ede06e70c35c11bf37791"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:25.283266Z","signature_b64":"jhhqC9Y/pqcWvxjBydYfgb2BVcUghxyVKTwm0eA3w36KXMILO/Yr8seTlyT0ozkfPFdOvxL3vb3wYjFp980hBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fcb47f099bb593cea983cce1dc528e8c8dd5f262cf99a61c35e2aab8f46f6b94","last_reissued_at":"2026-07-05T08:20:25.282791Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:25.282791Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SiDA-MoE: Sparsity-Inspired Data-Aware Serving for Efficient and Scalable Large Mixture-of-Experts Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Ang Li, Hai \"Helen\" Li, Jingwei Sun, Qilin Zheng, Shiyu Li, Xiangyu Jiang, Yiran Chen, Yongkai Wu, Yuhao Wu, Zhixu Du","submitted_at":"2023-10-29T01:08:55Z","abstract_excerpt":"Mixture-of-Experts (MoE) has emerged as a favorable architecture in the era of large models due to its inherent advantage, i.e., enlarging model capacity without incurring notable computational overhead. Yet, the realization of such benefits often results in ineffective GPU memory utilization, as large portions of the model parameters remain dormant during inference. Moreover, the memory demands of large models consistently outpace the memory capacity of contemporary GPUs. Addressing this, we introduce SiDA-MoE ($\\textbf{S}$parsity-$\\textbf{i}$nspired $\\textbf{D}$ata-$\\textbf{A}$ware), an effi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.18859","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.18859/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.18859","created_at":"2026-07-05T08:20:25.282866+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.18859v2","created_at":"2026-07-05T08:20:25.282866+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.18859","created_at":"2026-07-05T08:20:25.282866+00:00"},{"alias_kind":"pith_short_12","alias_value":"7S2H6CM3WWJ4","created_at":"2026-07-05T08:20:25.282866+00:00"},{"alias_kind":"pith_short_16","alias_value":"7S2H6CM3WWJ45KMD","created_at":"2026-07-05T08:20:25.282866+00:00"},{"alias_kind":"pith_short_8","alias_value":"7S2H6CM3","created_at":"2026-07-05T08:20:25.282866+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06888","citing_title":"Klotski: Efficient Mixture-of-Expert Inference via Expert-Aware Multi-Batch Pipeline","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS","json":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS.json","graph_json":"https://pith.science/api/pith-number/7S2H6CM3WWJ45KMDZTQ5YUUORS/graph.json","events_json":"https://pith.science/api/pith-number/7S2H6CM3WWJ45KMDZTQ5YUUORS/events.json","paper":"https://pith.science/paper/7S2H6CM3"},"agent_actions":{"view_html":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS","download_json":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS.json","view_paper":"https://pith.science/paper/7S2H6CM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.18859&json=true","fetch_graph":"https://pith.science/api/pith-number/7S2H6CM3WWJ45KMDZTQ5YUUORS/graph.json","fetch_events":"https://pith.science/api/pith-number/7S2H6CM3WWJ45KMDZTQ5YUUORS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS/action/storage_attestation","attest_author":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS/action/author_attestation","sign_citation":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS/action/citation_signature","submit_replication":"https://pith.science/pith/7S2H6CM3WWJ45KMDZTQ5YUUORS/action/replication_record"}},"created_at":"2026-07-05T08:20:25.282866+00:00","updated_at":"2026-07-05T08:20:25.282866+00:00"}