{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OZCNZ3MVHWX2SNMDYDGH5LM7OU","short_pith_number":"pith:OZCNZ3MV","schema_version":"1.0","canonical_sha256":"7644dced953dafa93583c0cc7ead9f752ad1e2874cf55dcc29728e510f73f200","source":{"kind":"arxiv","id":"2503.15921","version":1},"attestation_state":"computed","paper":{"title":"SPIN: Accelerating Large Language Model Inference with Heterogeneous Speculative Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Fahao Chen, Jing Deng, Peng Li, Tom H. Luan, Zhou Su","submitted_at":"2025-03-20T07:57:57Z","abstract_excerpt":"Speculative decoding has been shown as an effective way to accelerate Large Language Model (LLM) inference by using a Small Speculative Model (SSM) to generate candidate tokens in a so-called speculation phase, which are subsequently verified by the LLM in a verification phase. However, current state-of-the-art speculative decoding approaches have three key limitations: handling requests with varying difficulty using homogeneous SSMs, lack of robust support for batch processing, and insufficient holistic optimization for both speculation and verification phases. In this paper, we introduce SPI"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.15921","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-03-20T07:57:57Z","cross_cats_sorted":[],"title_canon_sha256":"8f3cde97e435ed36c8b05ce65b9dfeae903c7bea65fdbdfcace1aba4a5d28a60","abstract_canon_sha256":"5be2be38f8c5131d548a669d1215ca0ed77211745bffdd1026696303a489be58"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:36:01.835964Z","signature_b64":"zS6cHBZqTX98UurF3DijdhHJGEJFu444ViFUNCluOJdLKWzk2Xp6t//PvA0moO+78q3NNMelcg71SJDT/wwUDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7644dced953dafa93583c0cc7ead9f752ad1e2874cf55dcc29728e510f73f200","last_reissued_at":"2026-07-05T10:36:01.835506Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:36:01.835506Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SPIN: Accelerating Large Language Model Inference with Heterogeneous Speculative Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Fahao Chen, Jing Deng, Peng Li, Tom H. Luan, Zhou Su","submitted_at":"2025-03-20T07:57:57Z","abstract_excerpt":"Speculative decoding has been shown as an effective way to accelerate Large Language Model (LLM) inference by using a Small Speculative Model (SSM) to generate candidate tokens in a so-called speculation phase, which are subsequently verified by the LLM in a verification phase. However, current state-of-the-art speculative decoding approaches have three key limitations: handling requests with varying difficulty using homogeneous SSMs, lack of robust support for batch processing, and insufficient holistic optimization for both speculation and verification phases. In this paper, we introduce SPI"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.15921","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.15921/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.15921","created_at":"2026-07-05T10:36:01.835564+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.15921v1","created_at":"2026-07-05T10:36:01.835564+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.15921","created_at":"2026-07-05T10:36:01.835564+00:00"},{"alias_kind":"pith_short_12","alias_value":"OZCNZ3MVHWX2","created_at":"2026-07-05T10:36:01.835564+00:00"},{"alias_kind":"pith_short_16","alias_value":"OZCNZ3MVHWX2SNMD","created_at":"2026-07-05T10:36:01.835564+00:00"},{"alias_kind":"pith_short_8","alias_value":"OZCNZ3MV","created_at":"2026-07-05T10:36:01.835564+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24328","citing_title":"Speculative Verification: Exploiting Information Gain to Refine Speculative Decoding","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU","json":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU.json","graph_json":"https://pith.science/api/pith-number/OZCNZ3MVHWX2SNMDYDGH5LM7OU/graph.json","events_json":"https://pith.science/api/pith-number/OZCNZ3MVHWX2SNMDYDGH5LM7OU/events.json","paper":"https://pith.science/paper/OZCNZ3MV"},"agent_actions":{"view_html":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU","download_json":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU.json","view_paper":"https://pith.science/paper/OZCNZ3MV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.15921&json=true","fetch_graph":"https://pith.science/api/pith-number/OZCNZ3MVHWX2SNMDYDGH5LM7OU/graph.json","fetch_events":"https://pith.science/api/pith-number/OZCNZ3MVHWX2SNMDYDGH5LM7OU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU/action/storage_attestation","attest_author":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU/action/author_attestation","sign_citation":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU/action/citation_signature","submit_replication":"https://pith.science/pith/OZCNZ3MVHWX2SNMDYDGH5LM7OU/action/replication_record"}},"created_at":"2026-07-05T10:36:01.835564+00:00","updated_at":"2026-07-05T10:36:01.835564+00:00"}