{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XI6VZOPS3PS5VFGUJRKHDLUJHZ","short_pith_number":"pith:XI6VZOPS","schema_version":"1.0","canonical_sha256":"ba3d5cb9f2dbe5da94d44c5471ae893e465cce6543907c54a5ad585759239839","source":{"kind":"arxiv","id":"2502.04557","version":3},"attestation_state":"computed","paper":{"title":"Speeding up Speculative Decoding via Sequential Approximate Verification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Meiyu Zhong, Noel Teku, Ravi Tandon","submitted_at":"2025-02-06T23:10:53Z","abstract_excerpt":"Speculative Decoding (SD) is a recently proposed technique for faster inference using Large Language Models (LLMs). SD operates by using a smaller draft LLM for autoregressively generating a sequence of tokens and a larger target LLM for parallel verification to ensure statistical consistency. However, periodic parallel calls to the target LLM for verification prevent SD from achieving even lower latencies. We propose SPRINTER, which utilizes a low-complexity verifier trained to predict if tokens generated from a draft LLM would be accepted by the target LLM. By performing sequential approxima"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04557","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-06T23:10:53Z","cross_cats_sorted":["cs.IT","math.IT"],"title_canon_sha256":"096c206e9e1d1cfb4810dea53e95223e61ab945d751ec253c9c0a0693caa601c","abstract_canon_sha256":"234b9af5529132ff3330ae59311596b35c81103e05acd2c73635a6c7dd4f4188"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:19.625942Z","signature_b64":"Xr6Xwean0/EALVnnOyDdYmUIhUth61gFofs90trNA8aymtwV1/cU2ANsbA5sBpMsDXOCHTruBYCiR1ov/ngUBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba3d5cb9f2dbe5da94d44c5471ae893e465cce6543907c54a5ad585759239839","last_reissued_at":"2026-07-05T11:33:19.625464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:19.625464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speeding up Speculative Decoding via Sequential Approximate Verification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Meiyu Zhong, Noel Teku, Ravi Tandon","submitted_at":"2025-02-06T23:10:53Z","abstract_excerpt":"Speculative Decoding (SD) is a recently proposed technique for faster inference using Large Language Models (LLMs). SD operates by using a smaller draft LLM for autoregressively generating a sequence of tokens and a larger target LLM for parallel verification to ensure statistical consistency. However, periodic parallel calls to the target LLM for verification prevent SD from achieving even lower latencies. We propose SPRINTER, which utilizes a low-complexity verifier trained to predict if tokens generated from a draft LLM would be accepted by the target LLM. By performing sequential approxima"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04557","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04557/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04557","created_at":"2026-07-05T11:33:19.625523+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04557v3","created_at":"2026-07-05T11:33:19.625523+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04557","created_at":"2026-07-05T11:33:19.625523+00:00"},{"alias_kind":"pith_short_12","alias_value":"XI6VZOPS3PS5","created_at":"2026-07-05T11:33:19.625523+00:00"},{"alias_kind":"pith_short_16","alias_value":"XI6VZOPS3PS5VFGU","created_at":"2026-07-05T11:33:19.625523+00:00"},{"alias_kind":"pith_short_8","alias_value":"XI6VZOPS","created_at":"2026-07-05T11:33:19.625523+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30265","citing_title":"When Is a Draft Accepted? A Theory of Acceptance in Speculative Decoding","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ","json":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ.json","graph_json":"https://pith.science/api/pith-number/XI6VZOPS3PS5VFGUJRKHDLUJHZ/graph.json","events_json":"https://pith.science/api/pith-number/XI6VZOPS3PS5VFGUJRKHDLUJHZ/events.json","paper":"https://pith.science/paper/XI6VZOPS"},"agent_actions":{"view_html":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ","download_json":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ.json","view_paper":"https://pith.science/paper/XI6VZOPS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04557&json=true","fetch_graph":"https://pith.science/api/pith-number/XI6VZOPS3PS5VFGUJRKHDLUJHZ/graph.json","fetch_events":"https://pith.science/api/pith-number/XI6VZOPS3PS5VFGUJRKHDLUJHZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ/action/storage_attestation","attest_author":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ/action/author_attestation","sign_citation":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ/action/citation_signature","submit_replication":"https://pith.science/pith/XI6VZOPS3PS5VFGUJRKHDLUJHZ/action/replication_record"}},"created_at":"2026-07-05T11:33:19.625523+00:00","updated_at":"2026-07-05T11:33:19.625523+00:00"}