{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YQNSCRXNDLVT2XIVD7VQEJUKTZ","short_pith_number":"pith:YQNSCRXN","schema_version":"1.0","canonical_sha256":"c41b2146ed1aeb3d5d151feb02268a9e6a08eb08661eb29a6ccfb59fad1c3279","source":{"kind":"arxiv","id":"2508.19373","version":1},"attestation_state":"computed","paper":{"title":"HAP: Hybrid Adaptive Parallelism for Efficient Mixture-of-Experts Inference","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Han Bao, Haoran Lin, Kang Zhao, Ting Hu, Weiguo Liu, Wulong Liu, Xianzhi Yu, Xin Li, Zekun Yin, Zongyuan Zhan","submitted_at":"2025-08-26T19:07:52Z","abstract_excerpt":"Current inference systems for Mixture-of-Experts (MoE) models primarily employ static parallelization strategies. However, these static approaches cannot consistently achieve optimal performance across different inference scenarios, as they lack the flexibility to adapt to varying computational requirements. In this work, we propose HAP (Hybrid Adaptive Parallelism), a novel method that dynamically selects hybrid parallel strategies to enhance MoE inference efficiency. The fundamental innovation of HAP lies in hierarchically decomposing MoE architectures into two distinct computational modules"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.19373","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.DC","submitted_at":"2025-08-26T19:07:52Z","cross_cats_sorted":[],"title_canon_sha256":"6c49c783bf460be17714d8126108eb43e894ad0ffbd8dc22e7a6571575a23156","abstract_canon_sha256":"be3a22332b4cb287f471f943c15b0ca389189cc596f9fa868260d5e08864be27"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:44.366600Z","signature_b64":"zCC6lRsWv4Ii3Wi0QDsPOI2ac3KQVUjJzZ9nw2OcbfihA6cbQRjznBtsAHoln24/uOVZJeDau4ZJSwJNRMExBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c41b2146ed1aeb3d5d151feb02268a9e6a08eb08661eb29a6ccfb59fad1c3279","last_reissued_at":"2026-07-05T11:59:44.366087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:44.366087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HAP: Hybrid Adaptive Parallelism for Efficient Mixture-of-Experts Inference","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Han Bao, Haoran Lin, Kang Zhao, Ting Hu, Weiguo Liu, Wulong Liu, Xianzhi Yu, Xin Li, Zekun Yin, Zongyuan Zhan","submitted_at":"2025-08-26T19:07:52Z","abstract_excerpt":"Current inference systems for Mixture-of-Experts (MoE) models primarily employ static parallelization strategies. However, these static approaches cannot consistently achieve optimal performance across different inference scenarios, as they lack the flexibility to adapt to varying computational requirements. In this work, we propose HAP (Hybrid Adaptive Parallelism), a novel method that dynamically selects hybrid parallel strategies to enhance MoE inference efficiency. The fundamental innovation of HAP lies in hierarchically decomposing MoE architectures into two distinct computational modules"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.19373","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.19373/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.19373","created_at":"2026-07-05T11:59:44.366149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.19373v1","created_at":"2026-07-05T11:59:44.366149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.19373","created_at":"2026-07-05T11:59:44.366149+00:00"},{"alias_kind":"pith_short_12","alias_value":"YQNSCRXNDLVT","created_at":"2026-07-05T11:59:44.366149+00:00"},{"alias_kind":"pith_short_16","alias_value":"YQNSCRXNDLVT2XIV","created_at":"2026-07-05T11:59:44.366149+00:00"},{"alias_kind":"pith_short_8","alias_value":"YQNSCRXN","created_at":"2026-07-05T11:59:44.366149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26607","citing_title":"Moebius: Serving Mixture-of-Expert Models with Seamless Runtime Parallelism Switch","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ","json":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ.json","graph_json":"https://pith.science/api/pith-number/YQNSCRXNDLVT2XIVD7VQEJUKTZ/graph.json","events_json":"https://pith.science/api/pith-number/YQNSCRXNDLVT2XIVD7VQEJUKTZ/events.json","paper":"https://pith.science/paper/YQNSCRXN"},"agent_actions":{"view_html":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ","download_json":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ.json","view_paper":"https://pith.science/paper/YQNSCRXN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.19373&json=true","fetch_graph":"https://pith.science/api/pith-number/YQNSCRXNDLVT2XIVD7VQEJUKTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/YQNSCRXNDLVT2XIVD7VQEJUKTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ/action/storage_attestation","attest_author":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ/action/author_attestation","sign_citation":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ/action/citation_signature","submit_replication":"https://pith.science/pith/YQNSCRXNDLVT2XIVD7VQEJUKTZ/action/replication_record"}},"created_at":"2026-07-05T11:59:44.366149+00:00","updated_at":"2026-07-05T11:59:44.366149+00:00"}