{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IQXLJKCKODTCBEZVXNSWWQO27S","short_pith_number":"pith:IQXLJKCK","schema_version":"1.0","canonical_sha256":"442eb4a84a70e6209335bb656b41dafc8380a0c435b3635de205352f864bd20d","source":{"kind":"arxiv","id":"2410.08391","version":1},"attestation_state":"computed","paper":{"title":"KV Prediction for Improved Time to First Token","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenfan Sun, Maxwell Horton, Mohammad Rastegari, Moin Nabi, Qingqing Cao, Sachin Mehta, Yanzi Jin","submitted_at":"2024-10-10T21:55:11Z","abstract_excerpt":"Inference with transformer-based language models begins with a prompt processing step. In this step, the model generates the first output token and stores the KV cache needed for future generation steps. This prompt processing step can be computationally expensive, taking 10s of seconds or more for billion-parameter models on edge devices when prompt lengths or batch sizes rise. This degrades user experience by introducing significant latency into the model's outputs. To reduce the time spent producing the first output (known as the ``time to first token'', or TTFT) of a pretrained model, we i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08391","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-10T21:55:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ab1679719fb2c83eacba2ba6b019ac72243862645b3788b8f96241348aa589e9","abstract_canon_sha256":"359f2b8c3ae3220a068b66a968d1acbca72b401c201ab9fc605f2ddec2ed610e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:57.825025Z","signature_b64":"+BzqrnZsGKmsZrmY2Bei5Qy2wNljC+/3NDaExYpzoVBxGh60qzlI27BG72qNzMnjlg+Ii9Qr8+8iDQyzlQ15Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"442eb4a84a70e6209335bb656b41dafc8380a0c435b3635de205352f864bd20d","last_reissued_at":"2026-07-05T09:18:57.824496Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:57.824496Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KV Prediction for Improved Time to First Token","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenfan Sun, Maxwell Horton, Mohammad Rastegari, Moin Nabi, Qingqing Cao, Sachin Mehta, Yanzi Jin","submitted_at":"2024-10-10T21:55:11Z","abstract_excerpt":"Inference with transformer-based language models begins with a prompt processing step. In this step, the model generates the first output token and stores the KV cache needed for future generation steps. This prompt processing step can be computationally expensive, taking 10s of seconds or more for billion-parameter models on edge devices when prompt lengths or batch sizes rise. This degrades user experience by introducing significant latency into the model's outputs. To reduce the time spent producing the first output (known as the ``time to first token'', or TTFT) of a pretrained model, we i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08391","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08391/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08391","created_at":"2026-07-05T09:18:57.824577+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08391v1","created_at":"2026-07-05T09:18:57.824577+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08391","created_at":"2026-07-05T09:18:57.824577+00:00"},{"alias_kind":"pith_short_12","alias_value":"IQXLJKCKODTC","created_at":"2026-07-05T09:18:57.824577+00:00"},{"alias_kind":"pith_short_16","alias_value":"IQXLJKCKODTCBEZV","created_at":"2026-07-05T09:18:57.824577+00:00"},{"alias_kind":"pith_short_8","alias_value":"IQXLJKCK","created_at":"2026-07-05T09:18:57.824577+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08913","citing_title":"Non-Monotonic Latency in Apple MPS Decoding: KV Cache Interactions and Execution Regimes","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03314","citing_title":"When to Think, When to Speak: Learning Disclosure Policies for LLM Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08913","citing_title":"Non-Monotonic Latency in Apple MPS Decoding: KV Cache Interactions and Execution Regimes","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03314","citing_title":"When to Think, When to Speak: Learning Disclosure Policies for LLM Reasoning","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S","json":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S.json","graph_json":"https://pith.science/api/pith-number/IQXLJKCKODTCBEZVXNSWWQO27S/graph.json","events_json":"https://pith.science/api/pith-number/IQXLJKCKODTCBEZVXNSWWQO27S/events.json","paper":"https://pith.science/paper/IQXLJKCK"},"agent_actions":{"view_html":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S","download_json":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S.json","view_paper":"https://pith.science/paper/IQXLJKCK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08391&json=true","fetch_graph":"https://pith.science/api/pith-number/IQXLJKCKODTCBEZVXNSWWQO27S/graph.json","fetch_events":"https://pith.science/api/pith-number/IQXLJKCKODTCBEZVXNSWWQO27S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S/action/storage_attestation","attest_author":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S/action/author_attestation","sign_citation":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S/action/citation_signature","submit_replication":"https://pith.science/pith/IQXLJKCKODTCBEZVXNSWWQO27S/action/replication_record"}},"created_at":"2026-07-05T09:18:57.824577+00:00","updated_at":"2026-07-05T09:18:57.824577+00:00"}