{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7OAQI5DWNDKCA54P4TF2CSH5GU","short_pith_number":"pith:7OAQI5DW","schema_version":"1.0","canonical_sha256":"fb8104747668d420778fe4cba148fd35257abb31d78ca3182dd68bf009b4e08e","source":{"kind":"arxiv","id":"2502.02789","version":2},"attestation_state":"computed","paper":{"title":"Speculative Prefill: Turbocharging TTFT with Lightweight and Training-Free Token Importance Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Ce Zhang, Jingyu Liu","submitted_at":"2025-02-05T00:22:06Z","abstract_excerpt":"Improving time-to-first-token (TTFT) is an essentially important objective in modern large language model (LLM) inference engines. Optimizing TTFT directly results in higher maximal QPS and meets the requirements of many critical applications. However, boosting TTFT is notoriously challenging since it is compute-bounded and the performance bottleneck shifts from the self-attention that many prior works focus on to the MLP part. In this work, we present SpecPrefill, a training free framework that accelerates the inference TTFT for both long and medium context queries based on the following insi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02789","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-05T00:22:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dc7b78e83fa9a96807dfd2c10b9c8649c709720f50daf2cc16d01f2313ebaea3","abstract_canon_sha256":"73c7f7f4bf6f12c1117d81107f3e2cbcc2ab332787a5cbe47e3245626b0f47a7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:39.621344Z","signature_b64":"VqBrZ4CIjRz+iL2yCkcvmARbJTRqxmEU9fW5VhdspDdJwBGLC1RloHvPSLb0LR6MVxFmkS7jFberXSyt6whNAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb8104747668d420778fe4cba148fd35257abb31d78ca3182dd68bf009b4e08e","last_reissued_at":"2026-07-05T11:05:39.620895Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:39.620895Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speculative Prefill: Turbocharging TTFT with Lightweight and Training-Free Token Importance Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Ce Zhang, Jingyu Liu","submitted_at":"2025-02-05T00:22:06Z","abstract_excerpt":"Improving time-to-first-token (TTFT) is an essentially important objective in modern large language model (LLM) inference engines. Optimizing TTFT directly results in higher maximal QPS and meets the requirements of many critical applications. However, boosting TTFT is notoriously challenging since it is compute-bounded and the performance bottleneck shifts from the self-attention that many prior works focus on to the MLP part. In this work, we present SpecPrefill, a training free framework that accelerates the inference TTFT for both long and medium context queries based on the following insi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02789","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02789/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02789","created_at":"2026-07-05T11:05:39.620952+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02789v2","created_at":"2026-07-05T11:05:39.620952+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02789","created_at":"2026-07-05T11:05:39.620952+00:00"},{"alias_kind":"pith_short_12","alias_value":"7OAQI5DWNDKC","created_at":"2026-07-05T11:05:39.620952+00:00"},{"alias_kind":"pith_short_16","alias_value":"7OAQI5DWNDKCA54P","created_at":"2026-07-05T11:05:39.620952+00:00"},{"alias_kind":"pith_short_8","alias_value":"7OAQI5DW","created_at":"2026-07-05T11:05:39.620952+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12969","citing_title":"Multi-Modal Agents for Power Distribution Defect Detection: An Evaluation of Foundation Models","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU","json":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU.json","graph_json":"https://pith.science/api/pith-number/7OAQI5DWNDKCA54P4TF2CSH5GU/graph.json","events_json":"https://pith.science/api/pith-number/7OAQI5DWNDKCA54P4TF2CSH5GU/events.json","paper":"https://pith.science/paper/7OAQI5DW"},"agent_actions":{"view_html":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU","download_json":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU.json","view_paper":"https://pith.science/paper/7OAQI5DW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02789&json=true","fetch_graph":"https://pith.science/api/pith-number/7OAQI5DWNDKCA54P4TF2CSH5GU/graph.json","fetch_events":"https://pith.science/api/pith-number/7OAQI5DWNDKCA54P4TF2CSH5GU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU/action/storage_attestation","attest_author":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU/action/author_attestation","sign_citation":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU/action/citation_signature","submit_replication":"https://pith.science/pith/7OAQI5DWNDKCA54P4TF2CSH5GU/action/replication_record"}},"created_at":"2026-07-05T11:05:39.620952+00:00","updated_at":"2026-07-05T11:05:39.620952+00:00"}