{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RM2U7A2MEXJ6REXCOAAMML7L6V","short_pith_number":"pith:RM2U7A2M","schema_version":"1.0","canonical_sha256":"8b354f834c25d3e892e27000c62febf540dc00e3fc7f8c2944db0227a729203a","source":{"kind":"arxiv","id":"2501.05313","version":1},"attestation_state":"computed","paper":{"title":"Optimizing Distributed Deployment of Mixture-of-Experts Model Inference in Serverless Computing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Chuan Wu, Mengfan Liu, Wei Wang","submitted_at":"2025-01-09T15:29:33Z","abstract_excerpt":"With the advancement of serverless computing, running machine learning (ML) inference services over a serverless platform has been advocated, given its labor-free scalability and cost effectiveness. Mixture-of-Experts (MoE) models have been a dominant type of model architectures to enable large models nowadays, with parallel expert networks. Serving large MoE models on serverless computing is potentially beneficial, but has been underexplored due to substantial challenges in handling the skewed expert popularity and scatter-gather communication bottleneck in MoE model execution, for cost-effic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05313","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-01-09T15:29:33Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e62509acfb90e34dde10affe735ef4a4f4e515faf744a67b0a8c12fc773b9fa6","abstract_canon_sha256":"3110b57eddf080384c5c9f374d688394d18b1b863c9f521666c31afe42e3866f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:07.902453Z","signature_b64":"gNTnzw9AWNJZfxtPkKSCI8VmmmpHSjgwjdiKIJTdlrutTDMJyc4oX4+CPHUjXbSBUGKcMC3WK3Sq1Y/h+LR6Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b354f834c25d3e892e27000c62febf540dc00e3fc7f8c2944db0227a729203a","last_reissued_at":"2026-07-05T09:59:07.902014Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:07.902014Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimizing Distributed Deployment of Mixture-of-Experts Model Inference in Serverless Computing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Chuan Wu, Mengfan Liu, Wei Wang","submitted_at":"2025-01-09T15:29:33Z","abstract_excerpt":"With the advancement of serverless computing, running machine learning (ML) inference services over a serverless platform has been advocated, given its labor-free scalability and cost effectiveness. Mixture-of-Experts (MoE) models have been a dominant type of model architectures to enable large models nowadays, with parallel expert networks. Serving large MoE models on serverless computing is potentially beneficial, but has been underexplored due to substantial challenges in handling the skewed expert popularity and scatter-gather communication bottleneck in MoE model execution, for cost-effic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05313","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05313/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05313","created_at":"2026-07-05T09:59:07.902065+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05313v1","created_at":"2026-07-05T09:59:07.902065+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05313","created_at":"2026-07-05T09:59:07.902065+00:00"},{"alias_kind":"pith_short_12","alias_value":"RM2U7A2MEXJ6","created_at":"2026-07-05T09:59:07.902065+00:00"},{"alias_kind":"pith_short_16","alias_value":"RM2U7A2MEXJ6REXC","created_at":"2026-07-05T09:59:07.902065+00:00"},{"alias_kind":"pith_short_8","alias_value":"RM2U7A2M","created_at":"2026-07-05T09:59:07.902065+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.22919","citing_title":"Hecto: Modular Sparse Experts for Adaptive and Interpretable Reasoning","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V","json":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V.json","graph_json":"https://pith.science/api/pith-number/RM2U7A2MEXJ6REXCOAAMML7L6V/graph.json","events_json":"https://pith.science/api/pith-number/RM2U7A2MEXJ6REXCOAAMML7L6V/events.json","paper":"https://pith.science/paper/RM2U7A2M"},"agent_actions":{"view_html":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V","download_json":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V.json","view_paper":"https://pith.science/paper/RM2U7A2M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05313&json=true","fetch_graph":"https://pith.science/api/pith-number/RM2U7A2MEXJ6REXCOAAMML7L6V/graph.json","fetch_events":"https://pith.science/api/pith-number/RM2U7A2MEXJ6REXCOAAMML7L6V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V/action/storage_attestation","attest_author":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V/action/author_attestation","sign_citation":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V/action/citation_signature","submit_replication":"https://pith.science/pith/RM2U7A2MEXJ6REXCOAAMML7L6V/action/replication_record"}},"created_at":"2026-07-05T09:59:07.902065+00:00","updated_at":"2026-07-05T09:59:07.902065+00:00"}