{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZB2BIIQPB6JU2OOTXUQCDJMDCD","short_pith_number":"pith:ZB2BIIQP","schema_version":"1.0","canonical_sha256":"c87414220f0f934d39d3bd2021a58310f7c1e765e4a3741948a56a780dc8587a","source":{"kind":"arxiv","id":"2508.09591","version":1},"attestation_state":"computed","paper":{"title":"HierMoE: Accelerating MoE Training with Hierarchical Token Deduplication and Expert Swap","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Lin Zhang, Shaohuai Shi, Wenxiang Lin, Xiaowen Chu, Xinglin Pan, Xuan Wang","submitted_at":"2025-08-13T08:16:31Z","abstract_excerpt":"The sparsely activated mixture-of-experts (MoE) transformer has become a common architecture for large language models (LLMs) due to its sparsity, which requires fewer computational demands while easily scaling the model size. In MoE models, each MoE layer requires to dynamically choose tokens to activate particular experts for computation while the activated experts may not be located in the same device or GPU as the token. However, this leads to substantial communication and load imbalances across all GPUs, which obstructs the scalability of distributed systems within a GPU cluster. To this "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.09591","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-08-13T08:16:31Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ffd5b1a239779c676d33f6949592bc98d6282945a45a78ef8b3b77a4febd996a","abstract_canon_sha256":"28421b1bac3f5e07840fe0f7366485bc5a2ac3e06543dbd634cd4cdf73c6d721"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:17.807018Z","signature_b64":"COV3mdw8NXwDk1CXBT04DPbZG6yc1WhtsDv8h9ayanJ6uG+FZJv1v8bkx0DAvD8Bb6bYFbCx7ZOgQAs95cpMBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c87414220f0f934d39d3bd2021a58310f7c1e765e4a3741948a56a780dc8587a","last_reissued_at":"2026-07-05T11:53:17.806528Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:17.806528Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HierMoE: Accelerating MoE Training with Hierarchical Token Deduplication and Expert Swap","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Lin Zhang, Shaohuai Shi, Wenxiang Lin, Xiaowen Chu, Xinglin Pan, Xuan Wang","submitted_at":"2025-08-13T08:16:31Z","abstract_excerpt":"The sparsely activated mixture-of-experts (MoE) transformer has become a common architecture for large language models (LLMs) due to its sparsity, which requires fewer computational demands while easily scaling the model size. In MoE models, each MoE layer requires to dynamically choose tokens to activate particular experts for computation while the activated experts may not be located in the same device or GPU as the token. However, this leads to substantial communication and load imbalances across all GPUs, which obstructs the scalability of distributed systems within a GPU cluster. To this "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.09591","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.09591/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.09591","created_at":"2026-07-05T11:53:17.806580+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.09591v1","created_at":"2026-07-05T11:53:17.806580+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.09591","created_at":"2026-07-05T11:53:17.806580+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZB2BIIQPB6JU","created_at":"2026-07-05T11:53:17.806580+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZB2BIIQPB6JU2OOT","created_at":"2026-07-05T11:53:17.806580+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZB2BIIQP","created_at":"2026-07-05T11:53:17.806580+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06202","citing_title":"UBEP: Re-architecting Expert Parallelism Communication Library for Production Superpods","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06202","citing_title":"UBEP: Re-architecting Expert Parallelism Communication Library for Production Superpods","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2604.27844","citing_title":"ZipCCL: Efficient Lossless Data Compression of Communication Collectives for Accelerating LLM Training","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD","json":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD.json","graph_json":"https://pith.science/api/pith-number/ZB2BIIQPB6JU2OOTXUQCDJMDCD/graph.json","events_json":"https://pith.science/api/pith-number/ZB2BIIQPB6JU2OOTXUQCDJMDCD/events.json","paper":"https://pith.science/paper/ZB2BIIQP"},"agent_actions":{"view_html":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD","download_json":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD.json","view_paper":"https://pith.science/paper/ZB2BIIQP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.09591&json=true","fetch_graph":"https://pith.science/api/pith-number/ZB2BIIQPB6JU2OOTXUQCDJMDCD/graph.json","fetch_events":"https://pith.science/api/pith-number/ZB2BIIQPB6JU2OOTXUQCDJMDCD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD/action/storage_attestation","attest_author":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD/action/author_attestation","sign_citation":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD/action/citation_signature","submit_replication":"https://pith.science/pith/ZB2BIIQPB6JU2OOTXUQCDJMDCD/action/replication_record"}},"created_at":"2026-07-05T11:53:17.806580+00:00","updated_at":"2026-07-05T11:53:17.806580+00:00"}