{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4TQBR6TB2LWMZ6BLA7JQFKBLW6","short_pith_number":"pith:4TQBR6TB","schema_version":"1.0","canonical_sha256":"e4e018fa61d2ecccf82b07d302a82bb78a9874b236308717384d239ca1e962ac","source":{"kind":"arxiv","id":"2402.07033","version":3},"attestation_state":"computed","paper":{"title":"Fiddler: CPU-GPU Orchestration for Fast Inference of Mixture-of-Experts Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.OS"],"primary_cat":"cs.LG","authors_text":"Baris Kasikci, Kan Zhu, Keisuke Kamahori, Tian Tang, Yile Gu","submitted_at":"2024-02-10T19:54:08Z","abstract_excerpt":"Large Language Models (LLMs) with the Mixture-of-Experts (MoE) architectures have shown promising performance on various tasks. However, due to the huge model sizes, running them in resource-constrained environments where the GPU memory is not abundant is challenging. Some existing systems propose to use CPU resources to solve that, but they either suffer from the significant overhead of frequently moving data between CPU and GPU, or fail to consider distinct characteristics of CPUs and GPUs. This paper proposes Fiddler, a resource-efficient inference system for MoE models with limited GPU res"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07033","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-10T19:54:08Z","cross_cats_sorted":["cs.AI","cs.DC","cs.OS"],"title_canon_sha256":"6e0bfbfa6b733e822383d513bab3bf37f19cf54a980a21a67d1975d99a3fe56e","abstract_canon_sha256":"927c4391f7c4feddc73527dd7c9b06690c3c01825dc9b4bb1320c4ff42882f93"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:32.151165Z","signature_b64":"oSFUapgBMfl9iA56+59OMVH3jn61nmiu779vVy4r9gX71oo1D+yfd1cfnb3gFZplGrnEkcOtcMzGPlfnb8yBAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4e018fa61d2ecccf82b07d302a82bb78a9874b236308717384d239ca1e962ac","last_reissued_at":"2026-07-05T10:56:32.150676Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:32.150676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fiddler: CPU-GPU Orchestration for Fast Inference of Mixture-of-Experts Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.OS"],"primary_cat":"cs.LG","authors_text":"Baris Kasikci, Kan Zhu, Keisuke Kamahori, Tian Tang, Yile Gu","submitted_at":"2024-02-10T19:54:08Z","abstract_excerpt":"Large Language Models (LLMs) with the Mixture-of-Experts (MoE) architectures have shown promising performance on various tasks. However, due to the huge model sizes, running them in resource-constrained environments where the GPU memory is not abundant is challenging. Some existing systems propose to use CPU resources to solve that, but they either suffer from the significant overhead of frequently moving data between CPU and GPU, or fail to consider distinct characteristics of CPUs and GPUs. This paper proposes Fiddler, a resource-efficient inference system for MoE models with limited GPU res"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07033","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07033/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07033","created_at":"2026-07-05T10:56:32.150736+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07033v3","created_at":"2026-07-05T10:56:32.150736+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07033","created_at":"2026-07-05T10:56:32.150736+00:00"},{"alias_kind":"pith_short_12","alias_value":"4TQBR6TB2LWM","created_at":"2026-07-05T10:56:32.150736+00:00"},{"alias_kind":"pith_short_16","alias_value":"4TQBR6TB2LWMZ6BL","created_at":"2026-07-05T10:56:32.150736+00:00"},{"alias_kind":"pith_short_8","alias_value":"4TQBR6TB","created_at":"2026-07-05T10:56:32.150736+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21428","citing_title":"Does Mixture-of-Experts Actually Help Inference on Consumer and Edge Hardware? An Empirical Study","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10493","citing_title":"Achieving Cloud-Grade SLOs for Local Mixture-of-Experts Inference through CPU-GPU Hybrid Design","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29982","citing_title":"Beyond Uniform Experts: Cost-Aware Expert Execution for Efficient Multi-Device MoE Inference","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21198","citing_title":"ZipMoE: Efficient On-Device MoE Serving via Lossless Compression and Cache-Affinity Scheduling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2411.08982","citing_title":"Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17889","citing_title":"CoX-MoE: Coalesced Expert Execution for High-Throughput MoE Inference with AMX-Enabled CPU-GPU Co-Execution","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20179","citing_title":"TIDE: Efficient and Lossless MoE Diffusion LLM Inference with I/O-aware Expert Offload","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02960","citing_title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2508.12851","citing_title":"Accelerating Edge Inference for Distributed MoE Models with Latency-Optimized Expert Placement","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26074","citing_title":"DAK: Direct-Access-Enabled GPU Memory Offloading with Optimal Efficiency for LLM Inference","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05819","citing_title":"HCInfer: An Efficient Inference System via Error Compensation for Resource-Constrained Devices","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05899","citing_title":"VisMMOE: Exploiting Visual-Expert Affinity for Efficient Visual-Language MoE Offloading","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02960","citing_title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6","json":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6.json","graph_json":"https://pith.science/api/pith-number/4TQBR6TB2LWMZ6BLA7JQFKBLW6/graph.json","events_json":"https://pith.science/api/pith-number/4TQBR6TB2LWMZ6BLA7JQFKBLW6/events.json","paper":"https://pith.science/paper/4TQBR6TB"},"agent_actions":{"view_html":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6","download_json":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6.json","view_paper":"https://pith.science/paper/4TQBR6TB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07033&json=true","fetch_graph":"https://pith.science/api/pith-number/4TQBR6TB2LWMZ6BLA7JQFKBLW6/graph.json","fetch_events":"https://pith.science/api/pith-number/4TQBR6TB2LWMZ6BLA7JQFKBLW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6/action/storage_attestation","attest_author":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6/action/author_attestation","sign_citation":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6/action/citation_signature","submit_replication":"https://pith.science/pith/4TQBR6TB2LWMZ6BLA7JQFKBLW6/action/replication_record"}},"created_at":"2026-07-05T10:56:32.150736+00:00","updated_at":"2026-07-05T10:56:32.150736+00:00"}