{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M4L7DHMHQIODPLQGCNKJ2OP2YJ","short_pith_number":"pith:M4L7DHMH","schema_version":"1.0","canonical_sha256":"6717f19d87821c37ae0613549d39fac27a2fa84605b762d1517c050664257270","source":{"kind":"arxiv","id":"2408.10284","version":1},"attestation_state":"computed","paper":{"title":"AdapMoE: Adaptive Sensitivity-based Expert Gating and Management for Efficient MoE Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ling Liang, Meng Li, Ru Huang, Runsheng Wang, Shuzhang Zhong, Yuan Wang","submitted_at":"2024-08-19T03:27:15Z","abstract_excerpt":"Mixture-of-Experts (MoE) models are designed to enhance the efficiency of large language models (LLMs) without proportionally increasing the computational demands. However, their deployment on edge devices still faces significant challenges due to high on-demand loading overheads from managing sparsely activated experts. This paper introduces AdapMoE, an algorithm-system co-design framework for efficient MoE inference. AdapMoE features adaptive expert gating and management to reduce the on-demand loading overheads. We observe the heterogeneity of experts loading across layers and tokens, based"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.10284","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-08-19T03:27:15Z","cross_cats_sorted":[],"title_canon_sha256":"bf34bad876c452f15708ba9c28016b2db004cb27952cbd2655703a0f133d8e20","abstract_canon_sha256":"a35a73016c83c03f2cbee6a23ed78ebc035cb538b97e2cc9b085957bfa926a2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:09.175092Z","signature_b64":"Bg4L+H/9SjX/MLLhOEOavZqkuJlmJwpm5CBn9HdIFcqy7AyHG7pDqCJWCnpvhGq+EzsTS8u453FD10C15sWxDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6717f19d87821c37ae0613549d39fac27a2fa84605b762d1517c050664257270","last_reissued_at":"2026-07-05T08:57:09.174630Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:09.174630Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdapMoE: Adaptive Sensitivity-based Expert Gating and Management for Efficient MoE Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ling Liang, Meng Li, Ru Huang, Runsheng Wang, Shuzhang Zhong, Yuan Wang","submitted_at":"2024-08-19T03:27:15Z","abstract_excerpt":"Mixture-of-Experts (MoE) models are designed to enhance the efficiency of large language models (LLMs) without proportionally increasing the computational demands. However, their deployment on edge devices still faces significant challenges due to high on-demand loading overheads from managing sparsely activated experts. This paper introduces AdapMoE, an algorithm-system co-design framework for efficient MoE inference. AdapMoE features adaptive expert gating and management to reduce the on-demand loading overheads. We observe the heterogeneity of experts loading across layers and tokens, based"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.10284","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.10284/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.10284","created_at":"2026-07-05T08:57:09.174688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.10284v1","created_at":"2026-07-05T08:57:09.174688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.10284","created_at":"2026-07-05T08:57:09.174688+00:00"},{"alias_kind":"pith_short_12","alias_value":"M4L7DHMHQIOD","created_at":"2026-07-05T08:57:09.174688+00:00"},{"alias_kind":"pith_short_16","alias_value":"M4L7DHMHQIODPLQG","created_at":"2026-07-05T08:57:09.174688+00:00"},{"alias_kind":"pith_short_8","alias_value":"M4L7DHMH","created_at":"2026-07-05T08:57:09.174688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.12851","citing_title":"Accelerating Edge Inference for Distributed MoE Models with Latency-Optimized Expert Placement","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ","json":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ.json","graph_json":"https://pith.science/api/pith-number/M4L7DHMHQIODPLQGCNKJ2OP2YJ/graph.json","events_json":"https://pith.science/api/pith-number/M4L7DHMHQIODPLQGCNKJ2OP2YJ/events.json","paper":"https://pith.science/paper/M4L7DHMH"},"agent_actions":{"view_html":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ","download_json":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ.json","view_paper":"https://pith.science/paper/M4L7DHMH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.10284&json=true","fetch_graph":"https://pith.science/api/pith-number/M4L7DHMHQIODPLQGCNKJ2OP2YJ/graph.json","fetch_events":"https://pith.science/api/pith-number/M4L7DHMHQIODPLQGCNKJ2OP2YJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ/action/storage_attestation","attest_author":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ/action/author_attestation","sign_citation":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ/action/citation_signature","submit_replication":"https://pith.science/pith/M4L7DHMHQIODPLQGCNKJ2OP2YJ/action/replication_record"}},"created_at":"2026-07-05T08:57:09.174688+00:00","updated_at":"2026-07-05T08:57:09.174688+00:00"}