{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LA2LO2RUR2YXDHEGHNPDVIDLRM","short_pith_number":"pith:LA2LO2RU","schema_version":"1.0","canonical_sha256":"5834b76a348eb1719c863b5e3aa06b8b130e05d690c54c15f827cd222910b8be","source":{"kind":"arxiv","id":"2410.12247","version":2},"attestation_state":"computed","paper":{"title":"EPS-MoE: Expert Pipeline Scheduler for Cost-Efficient MoE Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CL","authors_text":"Fengcun Li, Jianchao Tan, Kefeng Zhang, Xiangyang Ji, Xiaoyu Zhao, Xunliang Cai, Yulei Qian","submitted_at":"2024-10-16T05:17:49Z","abstract_excerpt":"The Mixture-of-Experts (MoE) model has emerged as a prominent architecture in the field of Large Language Models (LLMs), providing a better balance between model performance and computational efficiency. However the General Matrix Multiply (GEMM) operations and large parameters introduce challenges related to computational efficiency and communication overhead, which become throughput bottlenecks during inference. Applying a single parallelism strategy like EP, DP, TP or a straightforward combination of them to MoE usually achieves sub-optimal inference throughput. This paper introduces EPS-Mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12247","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T05:17:49Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"eea0c782f3e7cd5bf02937d80377c7fd083deca6dabd32088da7dde597342178","abstract_canon_sha256":"9c556236abfa876e6d7bd3129e979d79c4e750185c690f7c85f1d930b083124a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:56:22.487188Z","signature_b64":"QCh6I6SYFQ7sfo1dROY+TwipxsGMRTgMjnP4nzikzN3v+144PGvO7h3UOUrSWHiAv1CmK94zrPVqBnUv4edkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5834b76a348eb1719c863b5e3aa06b8b130e05d690c54c15f827cd222910b8be","last_reissued_at":"2026-07-05T09:56:22.486688Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:56:22.486688Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EPS-MoE: Expert Pipeline Scheduler for Cost-Efficient MoE Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CL","authors_text":"Fengcun Li, Jianchao Tan, Kefeng Zhang, Xiangyang Ji, Xiaoyu Zhao, Xunliang Cai, Yulei Qian","submitted_at":"2024-10-16T05:17:49Z","abstract_excerpt":"The Mixture-of-Experts (MoE) model has emerged as a prominent architecture in the field of Large Language Models (LLMs), providing a better balance between model performance and computational efficiency. However the General Matrix Multiply (GEMM) operations and large parameters introduce challenges related to computational efficiency and communication overhead, which become throughput bottlenecks during inference. Applying a single parallelism strategy like EP, DP, TP or a straightforward combination of them to MoE usually achieves sub-optimal inference throughput. This paper introduces EPS-Mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12247","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12247/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12247","created_at":"2026-07-05T09:56:22.486750+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12247v2","created_at":"2026-07-05T09:56:22.486750+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12247","created_at":"2026-07-05T09:56:22.486750+00:00"},{"alias_kind":"pith_short_12","alias_value":"LA2LO2RUR2YX","created_at":"2026-07-05T09:56:22.486750+00:00"},{"alias_kind":"pith_short_16","alias_value":"LA2LO2RUR2YXDHEG","created_at":"2026-07-05T09:56:22.486750+00:00"},{"alias_kind":"pith_short_8","alias_value":"LA2LO2RU","created_at":"2026-07-05T09:56:22.486750+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.02960","citing_title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21613","citing_title":"Chameleon: Adaptive Fault Tolerance for Distributed Training via Real-time Policy Selection","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02960","citing_title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM","json":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM.json","graph_json":"https://pith.science/api/pith-number/LA2LO2RUR2YXDHEGHNPDVIDLRM/graph.json","events_json":"https://pith.science/api/pith-number/LA2LO2RUR2YXDHEGHNPDVIDLRM/events.json","paper":"https://pith.science/paper/LA2LO2RU"},"agent_actions":{"view_html":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM","download_json":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM.json","view_paper":"https://pith.science/paper/LA2LO2RU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12247&json=true","fetch_graph":"https://pith.science/api/pith-number/LA2LO2RUR2YXDHEGHNPDVIDLRM/graph.json","fetch_events":"https://pith.science/api/pith-number/LA2LO2RUR2YXDHEGHNPDVIDLRM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM/action/storage_attestation","attest_author":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM/action/author_attestation","sign_citation":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM/action/citation_signature","submit_replication":"https://pith.science/pith/LA2LO2RUR2YXDHEGHNPDVIDLRM/action/replication_record"}},"created_at":"2026-07-05T09:56:22.486750+00:00","updated_at":"2026-07-05T09:56:22.486750+00:00"}