{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PFEIO5ATYMLO326KJI3VWMSO7F","short_pith_number":"pith:PFEIO5AT","schema_version":"1.0","canonical_sha256":"7948877413c316edebca4a375b324ef969627d50d6fc614ec000671f74732714","source":{"kind":"arxiv","id":"2506.20675","version":1},"attestation_state":"computed","paper":{"title":"Utility-Driven Speculative Decoding for Mixture-of-Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Aamer Jaleel, Anish Saxena, Hritvik Taneja, Moinuddin Qureshi, Po-An Tsai","submitted_at":"2025-06-17T20:06:08Z","abstract_excerpt":"GPU memory bandwidth is the main bottleneck for low-latency Large Language Model (LLM) inference. Speculative decoding leverages idle GPU compute by using a lightweight drafter to propose K tokens, which the LLM verifies in parallel, boosting token throughput. In conventional dense LLMs, all model weights are fetched each iteration, so speculation adds no latency overhead. Emerging Mixture of Experts (MoE) models activate only a subset of weights per token, greatly reducing data movement. However, we show that speculation is ineffective for MoEs: draft tokens collectively activate more weights"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.20675","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-06-17T20:06:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"dd68d005fc4b3fce7442848fdef9ba477e03caf6ff23757113b06ff5df27a7af","abstract_canon_sha256":"e04b9c3b9b0bdfc7e51afd2fd4bd0ebd773b170151396773bc88ae135e7e1571"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:32.192662Z","signature_b64":"NftzfYLCyHcaenyeDqlhDjdN1MaAoFIXRHKRNd81Uxn2Ji33wA2JWz4Q8deDrYvQWYLaugQwbSEloDZkkdgdDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7948877413c316edebca4a375b324ef969627d50d6fc614ec000671f74732714","last_reissued_at":"2026-07-05T11:27:32.192001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:32.192001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Utility-Driven Speculative Decoding for Mixture-of-Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Aamer Jaleel, Anish Saxena, Hritvik Taneja, Moinuddin Qureshi, Po-An Tsai","submitted_at":"2025-06-17T20:06:08Z","abstract_excerpt":"GPU memory bandwidth is the main bottleneck for low-latency Large Language Model (LLM) inference. Speculative decoding leverages idle GPU compute by using a lightweight drafter to propose K tokens, which the LLM verifies in parallel, boosting token throughput. In conventional dense LLMs, all model weights are fetched each iteration, so speculation adds no latency overhead. Emerging Mixture of Experts (MoE) models activate only a subset of weights per token, greatly reducing data movement. However, we show that speculation is ineffective for MoEs: draft tokens collectively activate more weights"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20675","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.20675/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.20675","created_at":"2026-07-05T11:27:32.192064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.20675v1","created_at":"2026-07-05T11:27:32.192064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20675","created_at":"2026-07-05T11:27:32.192064+00:00"},{"alias_kind":"pith_short_12","alias_value":"PFEIO5ATYMLO","created_at":"2026-07-05T11:27:32.192064+00:00"},{"alias_kind":"pith_short_16","alias_value":"PFEIO5ATYMLO326K","created_at":"2026-07-05T11:27:32.192064+00:00"},{"alias_kind":"pith_short_8","alias_value":"PFEIO5AT","created_at":"2026-07-05T11:27:32.192064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00254","citing_title":"Rethinking Network Topologies for Cost-Effective Mixture-of-Experts LLM Serving","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00342","citing_title":"Making Every Verified Token Count: Adaptive Verification for MoE Speculative Decoding","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14626","citing_title":"ELMoE-3D: Leveraging Intrinsic Elasticity of MoE for Hybrid-Bonding-Enabled Self-Speculative Decoding in On-Premises Serving","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F","json":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F.json","graph_json":"https://pith.science/api/pith-number/PFEIO5ATYMLO326KJI3VWMSO7F/graph.json","events_json":"https://pith.science/api/pith-number/PFEIO5ATYMLO326KJI3VWMSO7F/events.json","paper":"https://pith.science/paper/PFEIO5AT"},"agent_actions":{"view_html":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F","download_json":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F.json","view_paper":"https://pith.science/paper/PFEIO5AT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.20675&json=true","fetch_graph":"https://pith.science/api/pith-number/PFEIO5ATYMLO326KJI3VWMSO7F/graph.json","fetch_events":"https://pith.science/api/pith-number/PFEIO5ATYMLO326KJI3VWMSO7F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F/action/storage_attestation","attest_author":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F/action/author_attestation","sign_citation":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F/action/citation_signature","submit_replication":"https://pith.science/pith/PFEIO5ATYMLO326KJI3VWMSO7F/action/replication_record"}},"created_at":"2026-07-05T11:27:32.192064+00:00","updated_at":"2026-07-05T11:27:32.192064+00:00"}