{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NM3GVWSFEVNJ2BKP4PWPYTBSPI","short_pith_number":"pith:NM3GVWSF","schema_version":"1.0","canonical_sha256":"6b366ada45255a9d054fe3ecfc4c327a0833871384b1ba57ac9cfb583e41513a","source":{"kind":"arxiv","id":"2410.06916","version":2},"attestation_state":"computed","paper":{"title":"SWIFT: On-the-Fly Self-Speculative Decoding for LLM Inference Acceleration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cunxiao Du, Heming Xia, Jun Zhang, Wenjie Li, Yongqi Li","submitted_at":"2024-10-09T14:15:30Z","abstract_excerpt":"Speculative decoding (SD) has emerged as a widely used paradigm to accelerate LLM inference without compromising quality. It works by first employing a compact model to draft multiple tokens efficiently and then using the target LLM to verify them in parallel. While this technique has achieved notable speedups, most existing approaches necessitate either additional parameters or extensive training to construct effective draft models, thereby restricting their applicability across different LLMs and tasks. To address this limitation, we explore a novel plug-and-play SD solution with layer-skipp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.06916","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-09T14:15:30Z","cross_cats_sorted":[],"title_canon_sha256":"019fd6520ce77a5c6f3eae2dc0d2247c73232af01819cd329a15d7a26919e9fa","abstract_canon_sha256":"011e9955422f5833cddadccbd2d421036c4cf34d0aafb6d8b62d8a4fba2df741"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:11.518862Z","signature_b64":"itgO/sRTDiHO8d98qWQicWsIRJW8Lgy1TYCnboRpTC39lsM06Q4Rl5jKuA1ZxXx+qoLMT8mZyUcVU0OhwURwBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b366ada45255a9d054fe3ecfc4c327a0833871384b1ba57ac9cfb583e41513a","last_reissued_at":"2026-07-05T10:25:11.518283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:11.518283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SWIFT: On-the-Fly Self-Speculative Decoding for LLM Inference Acceleration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cunxiao Du, Heming Xia, Jun Zhang, Wenjie Li, Yongqi Li","submitted_at":"2024-10-09T14:15:30Z","abstract_excerpt":"Speculative decoding (SD) has emerged as a widely used paradigm to accelerate LLM inference without compromising quality. It works by first employing a compact model to draft multiple tokens efficiently and then using the target LLM to verify them in parallel. While this technique has achieved notable speedups, most existing approaches necessitate either additional parameters or extensive training to construct effective draft models, thereby restricting their applicability across different LLMs and tasks. To address this limitation, we explore a novel plug-and-play SD solution with layer-skipp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.06916","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.06916/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.06916","created_at":"2026-07-05T10:25:11.518354+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.06916v2","created_at":"2026-07-05T10:25:11.518354+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06916","created_at":"2026-07-05T10:25:11.518354+00:00"},{"alias_kind":"pith_short_12","alias_value":"NM3GVWSFEVNJ","created_at":"2026-07-05T10:25:11.518354+00:00"},{"alias_kind":"pith_short_16","alias_value":"NM3GVWSFEVNJ2BKP","created_at":"2026-07-05T10:25:11.518354+00:00"},{"alias_kind":"pith_short_8","alias_value":"NM3GVWSF","created_at":"2026-07-05T10:25:11.518354+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26558","citing_title":"Cassandra: Enabling Reasoning LLMs at Edge via Self-Speculative Decoding","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11567","citing_title":"Dynamic Execution Commitment of Vision-Language-Action Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29223","citing_title":"Depth Exploration for LLM Decoding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11567","citing_title":"Dynamic Execution Commitment of Vision-Language-Action Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11567","citing_title":"Dynamic Execution Commitment of Vision-Language-Action Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26412","citing_title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26412","citing_title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01106","citing_title":"Component-Aware Self-Speculative Decoding in Hybrid Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20503","citing_title":"FASER: Fine-Grained Phase Management for Speculative Decoding in Dynamic LLM Serving","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI","json":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI.json","graph_json":"https://pith.science/api/pith-number/NM3GVWSFEVNJ2BKP4PWPYTBSPI/graph.json","events_json":"https://pith.science/api/pith-number/NM3GVWSFEVNJ2BKP4PWPYTBSPI/events.json","paper":"https://pith.science/paper/NM3GVWSF"},"agent_actions":{"view_html":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI","download_json":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI.json","view_paper":"https://pith.science/paper/NM3GVWSF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.06916&json=true","fetch_graph":"https://pith.science/api/pith-number/NM3GVWSFEVNJ2BKP4PWPYTBSPI/graph.json","fetch_events":"https://pith.science/api/pith-number/NM3GVWSFEVNJ2BKP4PWPYTBSPI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI/action/storage_attestation","attest_author":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI/action/author_attestation","sign_citation":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI/action/citation_signature","submit_replication":"https://pith.science/pith/NM3GVWSFEVNJ2BKP4PWPYTBSPI/action/replication_record"}},"created_at":"2026-07-05T10:25:11.518354+00:00","updated_at":"2026-07-05T10:25:11.518354+00:00"}