{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AAAPBCEIMR3IMNVTKKP2JNJY3V","short_pith_number":"pith:AAAPBCEI","schema_version":"1.0","canonical_sha256":"0000f0888864768636b3529fa4b538dd5a8ff3037333d27f9589f732e32bbf39","source":{"kind":"arxiv","id":"2308.14352","version":2},"attestation_state":"computed","paper":{"title":"EdgeMoE: Empowering Sparse Large Language Models on Mobile Devices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ao Zhou, Liwei Guo, Mengwei Xu, Rongjie Yi, Shangguang Wang, Shiyun Wei","submitted_at":"2023-08-28T06:56:08Z","abstract_excerpt":"Large language models (LLMs) such as GPTs and Mixtral-8x7B have revolutionized machine intelligence due to their exceptional abilities in generic ML tasks. Transiting LLMs from datacenters to edge devices brings benefits like better privacy and availability, but is challenged by their massive parameter size and thus unbearable runtime costs. To this end, we present EdgeMoE, an on-device inference engine for mixture-of-expert (MoE) LLMs -- a popular form of sparse LLM that scales its parameter size with almost constant computing complexity. EdgeMoE achieves both memory- and compute-efficiency b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.14352","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-28T06:56:08Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"3e1fb32bc6993845392126b5b103ef023e6e180fc2a9e931ea6999a5375591f8","abstract_canon_sha256":"3509af7b0f6cb8658fb80021b952e49d7e082fc1c3c0338b236d47deccd150cc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:59.067407Z","signature_b64":"fvUlTYvpohIFAIzIoRGBTwPo6h7FsnphlhaeFKqGjEUsyDzKNljWqurcxhv+jIDsx7VLqrTxNBNv0A/YA6o4AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0000f0888864768636b3529fa4b538dd5a8ff3037333d27f9589f732e32bbf39","last_reissued_at":"2026-07-05T10:25:59.066827Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:59.066827Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EdgeMoE: Empowering Sparse Large Language Models on Mobile Devices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ao Zhou, Liwei Guo, Mengwei Xu, Rongjie Yi, Shangguang Wang, Shiyun Wei","submitted_at":"2023-08-28T06:56:08Z","abstract_excerpt":"Large language models (LLMs) such as GPTs and Mixtral-8x7B have revolutionized machine intelligence due to their exceptional abilities in generic ML tasks. Transiting LLMs from datacenters to edge devices brings benefits like better privacy and availability, but is challenged by their massive parameter size and thus unbearable runtime costs. To this end, we present EdgeMoE, an on-device inference engine for mixture-of-expert (MoE) LLMs -- a popular form of sparse LLM that scales its parameter size with almost constant computing complexity. EdgeMoE achieves both memory- and compute-efficiency b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.14352","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.14352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.14352","created_at":"2026-07-05T10:25:59.066898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.14352v2","created_at":"2026-07-05T10:25:59.066898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.14352","created_at":"2026-07-05T10:25:59.066898+00:00"},{"alias_kind":"pith_short_12","alias_value":"AAAPBCEIMR3I","created_at":"2026-07-05T10:25:59.066898+00:00"},{"alias_kind":"pith_short_16","alias_value":"AAAPBCEIMR3IMNVT","created_at":"2026-07-05T10:25:59.066898+00:00"},{"alias_kind":"pith_short_8","alias_value":"AAAPBCEI","created_at":"2026-07-05T10:25:59.066898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23294","citing_title":"NASiC: 3D NAND-based CAM-Selected Multibit CIM Architecture for Efficient On-Device Mixture-of-Experts LLM Inference","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2508.12851","citing_title":"Accelerating Edge Inference for Distributed MoE Models with Latency-Optimized Expert Placement","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07379","citing_title":"DuoServe-MoE: Dual-Phase Expert Prefetch and Caching for LLM Inference QoS Assurance","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V","json":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V.json","graph_json":"https://pith.science/api/pith-number/AAAPBCEIMR3IMNVTKKP2JNJY3V/graph.json","events_json":"https://pith.science/api/pith-number/AAAPBCEIMR3IMNVTKKP2JNJY3V/events.json","paper":"https://pith.science/paper/AAAPBCEI"},"agent_actions":{"view_html":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V","download_json":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V.json","view_paper":"https://pith.science/paper/AAAPBCEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.14352&json=true","fetch_graph":"https://pith.science/api/pith-number/AAAPBCEIMR3IMNVTKKP2JNJY3V/graph.json","fetch_events":"https://pith.science/api/pith-number/AAAPBCEIMR3IMNVTKKP2JNJY3V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V/action/storage_attestation","attest_author":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V/action/author_attestation","sign_citation":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V/action/citation_signature","submit_replication":"https://pith.science/pith/AAAPBCEIMR3IMNVTKKP2JNJY3V/action/replication_record"}},"created_at":"2026-07-05T10:25:59.066898+00:00","updated_at":"2026-07-05T10:25:59.066898+00:00"}