{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6ETBXRXWREPDGHKW6WCCONGJ2Z","short_pith_number":"pith:6ETBXRXW","schema_version":"1.0","canonical_sha256":"f1261bc6f6891e331d56f5842734c9d6789ea93de3ccf70444e7d854f4f29c0b","source":{"kind":"arxiv","id":"2504.15299","version":1},"attestation_state":"computed","paper":{"title":"D$^{2}$MoE: Dual Routing and Dynamic Scheduling for Efficient On-Device MoE-based LLM Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Haodong Wang, Qihua Zhou, Song Guo, Zicong Hong","submitted_at":"2025-04-17T05:37:35Z","abstract_excerpt":"The mixture of experts (MoE) model is a sparse variant of large language models (LLMs), designed to hold a better balance between intelligent capability and computational overhead. Despite its benefits, MoE is still too expensive to deploy on resource-constrained edge devices, especially with the demands of on-device inference services. Recent research efforts often apply model compression techniques, such as quantization, pruning and merging, to restrict MoE complexity. Unfortunately, due to their predefined static model optimization strategies, they cannot always achieve the desired quality-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.15299","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-04-17T05:37:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e765e6295abcbd67612eeea4f777bd02fbfee19e82b7388dcb20da69ccddfe9c","abstract_canon_sha256":"cfda398aee1a5398c9cd470f96957c6b9b155ffdcb8a14294e50d619ffe4fe49"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:00.144758Z","signature_b64":"rRjaJGmIxBTXuBrdARWsAJj+ajEgyHaWEbQLCQWcMp7GiTJ+Wa8beIslPpAUoS096pG7T8Jt9LxOJaiM0mi8BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1261bc6f6891e331d56f5842734c9d6789ea93de3ccf70444e7d854f4f29c0b","last_reissued_at":"2026-07-05T10:52:00.144248Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:00.144248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"D$^{2}$MoE: Dual Routing and Dynamic Scheduling for Efficient On-Device MoE-based LLM Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Haodong Wang, Qihua Zhou, Song Guo, Zicong Hong","submitted_at":"2025-04-17T05:37:35Z","abstract_excerpt":"The mixture of experts (MoE) model is a sparse variant of large language models (LLMs), designed to hold a better balance between intelligent capability and computational overhead. Despite its benefits, MoE is still too expensive to deploy on resource-constrained edge devices, especially with the demands of on-device inference services. Recent research efforts often apply model compression techniques, such as quantization, pruning and merging, to restrict MoE complexity. Unfortunately, due to their predefined static model optimization strategies, they cannot always achieve the desired quality-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.15299","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.15299/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.15299","created_at":"2026-07-05T10:52:00.144308+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.15299v1","created_at":"2026-07-05T10:52:00.144308+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.15299","created_at":"2026-07-05T10:52:00.144308+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ETBXRXWREPD","created_at":"2026-07-05T10:52:00.144308+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ETBXRXWREPDGHKW","created_at":"2026-07-05T10:52:00.144308+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ETBXRXW","created_at":"2026-07-05T10:52:00.144308+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z","json":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z.json","graph_json":"https://pith.science/api/pith-number/6ETBXRXWREPDGHKW6WCCONGJ2Z/graph.json","events_json":"https://pith.science/api/pith-number/6ETBXRXWREPDGHKW6WCCONGJ2Z/events.json","paper":"https://pith.science/paper/6ETBXRXW"},"agent_actions":{"view_html":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z","download_json":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z.json","view_paper":"https://pith.science/paper/6ETBXRXW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.15299&json=true","fetch_graph":"https://pith.science/api/pith-number/6ETBXRXWREPDGHKW6WCCONGJ2Z/graph.json","fetch_events":"https://pith.science/api/pith-number/6ETBXRXWREPDGHKW6WCCONGJ2Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z/action/storage_attestation","attest_author":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z/action/author_attestation","sign_citation":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z/action/citation_signature","submit_replication":"https://pith.science/pith/6ETBXRXWREPDGHKW6WCCONGJ2Z/action/replication_record"}},"created_at":"2026-07-05T10:52:00.144308+00:00","updated_at":"2026-07-05T10:52:00.144308+00:00"}