{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IWFB3USPJA4FCMFZ6W2L7NS6GE","short_pith_number":"pith:IWFB3USP","schema_version":"1.0","canonical_sha256":"458a1dd24f48385130b9f5b4bfb65e31255a5330f6d0fe2af5a1fe2c731ab5ef","source":{"kind":"arxiv","id":"2502.06643","version":1},"attestation_state":"computed","paper":{"title":"MoETuner: Optimized Mixture of Expert Serving with Balanced Expert Placement and Token Routing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Divya Mahajan, Seokjin Go","submitted_at":"2025-02-10T16:34:36Z","abstract_excerpt":"Mixture-of-Experts (MoE) model architecture has emerged as a promising solution for scaling transformer models efficiently, offering sparse activation that reduces computational costs while increasing model capacity. However, as MoE models scale, they need to be distributed across GPU devices, thus face critical performance bottlenecks due to their large memory footprint. Expert parallelism distributes experts across GPUs, however, faces key challenges including an unbalanced token routing and expert activation, resulting in communication tail latency and processing inefficiencies. While exist"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06643","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T16:34:36Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"3727a9803775c3d1860c545c5aba2a80fef6bdd9becb1c5a1142a969929b0329","abstract_canon_sha256":"98891c9e3a993f68e48280dec42a48a3f89602383120ddea41cd1e70213e2395"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:10.032848Z","signature_b64":"DZrsV3m2zXC+/Yin7F4Gsl4+6YR2iyTOD9zpXaY1L5tJejQBiBa1rBOlSAAJKLdYOIwoZLQBPkPOUOCuEKkuDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"458a1dd24f48385130b9f5b4bfb65e31255a5330f6d0fe2af5a1fe2c731ab5ef","last_reissued_at":"2026-07-05T10:12:10.032399Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:10.032399Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoETuner: Optimized Mixture of Expert Serving with Balanced Expert Placement and Token Routing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Divya Mahajan, Seokjin Go","submitted_at":"2025-02-10T16:34:36Z","abstract_excerpt":"Mixture-of-Experts (MoE) model architecture has emerged as a promising solution for scaling transformer models efficiently, offering sparse activation that reduces computational costs while increasing model capacity. However, as MoE models scale, they need to be distributed across GPU devices, thus face critical performance bottlenecks due to their large memory footprint. Expert parallelism distributes experts across GPUs, however, faces key challenges including an unbalanced token routing and expert activation, resulting in communication tail latency and processing inefficiencies. While exist"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06643","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06643/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06643","created_at":"2026-07-05T10:12:10.032450+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06643v1","created_at":"2026-07-05T10:12:10.032450+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06643","created_at":"2026-07-05T10:12:10.032450+00:00"},{"alias_kind":"pith_short_12","alias_value":"IWFB3USPJA4F","created_at":"2026-07-05T10:12:10.032450+00:00"},{"alias_kind":"pith_short_16","alias_value":"IWFB3USPJA4FCMFZ","created_at":"2026-07-05T10:12:10.032450+00:00"},{"alias_kind":"pith_short_8","alias_value":"IWFB3USP","created_at":"2026-07-05T10:12:10.032450+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00466","citing_title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00466","citing_title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01007","citing_title":"Beyond Task-Agnostic: Task-Aware Grouping for Communication-Efficient Multi-Task MoE Inference","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00735","citing_title":"ViBE: Co-Optimizing Workload Skew and Hardware Variability for MoE Serving","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26845","citing_title":"Birkhoff Decompositions and Photonic Interconnects Wait! Don't Forget the Compute!","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00515","citing_title":"SpaceMoE: Realizing Distributed Mixture-of-Experts Inference over Space Networks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20982","citing_title":"Diagnosing Overhead in Dispatch Operations: Cross-architecture Observatory","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21100","citing_title":"NanoCP: Request-Level Dynamic Context Parallelism for Data-Expert Parallel Decoding","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19945","citing_title":"GEM: GPU-Variability-Aware Expert to GPU Mapping for MoE Systems","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2508.12851","citing_title":"Accelerating Edge Inference for Distributed MoE Models with Latency-Optimized Expert Placement","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25041","citing_title":"GRACE-MoE: Grouping and Replication with Locality-Aware Routing for Efficient Distributed MoE Inference","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05497","citing_title":"Patterns behind Chaos: Forecasting Data Movement for Efficient Large-Scale MoE LLM Inference","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08292","citing_title":"Hierarchical Mixture-of-Experts with Two-Stage Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23150","citing_title":"Scaling Multi-Node Mixture-of-Experts Inference Using Expert Activation Patterns","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06206","citing_title":"Federation of Experts: Communication Efficient Distributed Inference for Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00515","citing_title":"SpaceMoE: Realizing Distributed Mixture-of-Experts Inference over Space Networks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00209","citing_title":"Replication in Graph Partitioning and Scheduling Problems","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE","json":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE.json","graph_json":"https://pith.science/api/pith-number/IWFB3USPJA4FCMFZ6W2L7NS6GE/graph.json","events_json":"https://pith.science/api/pith-number/IWFB3USPJA4FCMFZ6W2L7NS6GE/events.json","paper":"https://pith.science/paper/IWFB3USP"},"agent_actions":{"view_html":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE","download_json":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE.json","view_paper":"https://pith.science/paper/IWFB3USP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06643&json=true","fetch_graph":"https://pith.science/api/pith-number/IWFB3USPJA4FCMFZ6W2L7NS6GE/graph.json","fetch_events":"https://pith.science/api/pith-number/IWFB3USPJA4FCMFZ6W2L7NS6GE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE/action/storage_attestation","attest_author":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE/action/author_attestation","sign_citation":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE/action/citation_signature","submit_replication":"https://pith.science/pith/IWFB3USPJA4FCMFZ6W2L7NS6GE/action/replication_record"}},"created_at":"2026-07-05T10:12:10.032450+00:00","updated_at":"2026-07-05T10:12:10.032450+00:00"}