{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:COPO5K72LFKIP422K7V25PBSOU","short_pith_number":"pith:COPO5K72","schema_version":"1.0","canonical_sha256":"139eeeabfa595487f35a57ebaebc327515d57aa358da5a4d125394f7e2afe706","source":{"kind":"arxiv","id":"2406.02532","version":3},"attestation_state":"computed","paper":{"title":"SpecExec: Massively Parallel Speculative Decoding for Interactive LLM Inference on Consumer Devices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Avner May, Beidi Chen, Max Ryabinin, Ruslan Svirschevski, Zhihao Jia, Zhuoming Chen","submitted_at":"2024-06-04T17:53:36Z","abstract_excerpt":"As large language models gain widespread adoption, running them efficiently becomes crucial. Recent works on LLM inference use speculative decoding to achieve extreme speedups. However, most of these works implicitly design their algorithms for high-end datacenter hardware. In this work, we ask the opposite question: how fast can we run LLMs on consumer machines? Consumer GPUs can no longer fit the largest available models (50B+ parameters) and must offload them to RAM or SSD. When running with offloaded parameters, the inference engine can process batches of hundreds or thousands of tokens at"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02532","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-04T17:53:36Z","cross_cats_sorted":[],"title_canon_sha256":"c5decb4ba72e8b6d280950fb13c3f8cab9114c520c15f0243f68aee75ea1d445","abstract_canon_sha256":"15e830d680d6ede9466529c271ebf6ff5718db99bd380d01e3066ec2ac952e99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:25.977149Z","signature_b64":"7ry+tLsXkYOCN1pCycbyNtkqWkQ1602XAZooBqSRJbHtLyKKZf/i7z00OjaxkLLQH0Vo7vTix9hrYmcG0K1zAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"139eeeabfa595487f35a57ebaebc327515d57aa358da5a4d125394f7e2afe706","last_reissued_at":"2026-07-05T09:42:25.976613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:25.976613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpecExec: Massively Parallel Speculative Decoding for Interactive LLM Inference on Consumer Devices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Avner May, Beidi Chen, Max Ryabinin, Ruslan Svirschevski, Zhihao Jia, Zhuoming Chen","submitted_at":"2024-06-04T17:53:36Z","abstract_excerpt":"As large language models gain widespread adoption, running them efficiently becomes crucial. Recent works on LLM inference use speculative decoding to achieve extreme speedups. However, most of these works implicitly design their algorithms for high-end datacenter hardware. In this work, we ask the opposite question: how fast can we run LLMs on consumer machines? Consumer GPUs can no longer fit the largest available models (50B+ parameters) and must offload them to RAM or SSD. When running with offloaded parameters, the inference engine can process batches of hundreds or thousands of tokens at"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02532","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02532","created_at":"2026-07-05T09:42:25.976674+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02532v3","created_at":"2026-07-05T09:42:25.976674+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02532","created_at":"2026-07-05T09:42:25.976674+00:00"},{"alias_kind":"pith_short_12","alias_value":"COPO5K72LFKI","created_at":"2026-07-05T09:42:25.976674+00:00"},{"alias_kind":"pith_short_16","alias_value":"COPO5K72LFKIP422","created_at":"2026-07-05T09:42:25.976674+00:00"},{"alias_kind":"pith_short_8","alias_value":"COPO5K72","created_at":"2026-07-05T09:42:25.976674+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12303","citing_title":"From 2D Grids to 1D Tokens: Reforming Shared Representations for Multimodal Image Fusion","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20503","citing_title":"FASER: Fine-Grained Phase Management for Speculative Decoding in Dynamic LLM Serving","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU","json":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU.json","graph_json":"https://pith.science/api/pith-number/COPO5K72LFKIP422K7V25PBSOU/graph.json","events_json":"https://pith.science/api/pith-number/COPO5K72LFKIP422K7V25PBSOU/events.json","paper":"https://pith.science/paper/COPO5K72"},"agent_actions":{"view_html":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU","download_json":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU.json","view_paper":"https://pith.science/paper/COPO5K72","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02532&json=true","fetch_graph":"https://pith.science/api/pith-number/COPO5K72LFKIP422K7V25PBSOU/graph.json","fetch_events":"https://pith.science/api/pith-number/COPO5K72LFKIP422K7V25PBSOU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU/action/storage_attestation","attest_author":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU/action/author_attestation","sign_citation":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU/action/citation_signature","submit_replication":"https://pith.science/pith/COPO5K72LFKIP422K7V25PBSOU/action/replication_record"}},"created_at":"2026-07-05T09:42:25.976674+00:00","updated_at":"2026-07-05T09:42:25.976674+00:00"}