{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RVRUQCV6X7V7BY7F6IJYKZPU7X","short_pith_number":"pith:RVRUQCV6","schema_version":"1.0","canonical_sha256":"8d63480abebfebf0e3e5f2138565f4fdc54ecd434a3ab7fe91c8bfa49b233ef5","source":{"kind":"arxiv","id":"2307.05908","version":2},"attestation_state":"computed","paper":{"title":"Predictive Pipelined Decoding: A Compute-Latency Trade-off for Exact LLM Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dimitris Papailiopoulos, Gibbeum Lee, Jaewoong Cho, Kangwook Lee, Seongjun Yang","submitted_at":"2023-07-12T04:28:41Z","abstract_excerpt":"This paper presents \"Predictive Pipelined Decoding (PPD),\" an approach that speeds up greedy decoding in Large Language Models (LLMs) while maintaining the exact same output as the original decoding. Unlike conventional strategies, PPD employs additional compute resources to parallelize the initiation of subsequent token decoding during the current token decoding. This method reduces decoding latency and reshapes the understanding of trade-offs in LLM decoding strategies. We have developed a theoretical framework that allows us to analyze the trade-off between computation and latency. Using th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.05908","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-12T04:28:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d710f281ebae610c05e9002c9f00e320e8d6a7e7ebe4d18a4dcd89f1909973b1","abstract_canon_sha256":"afb76197f0dd29a0bb7bde3525b0f0048b0f3b3cec21ecb1e11d210859e627bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:25.024077Z","signature_b64":"q0rNcVSXmaitxt5fbww0YKza6t1pd8Ukud41GEfQlezn5hLrS0ImzUNoNgLMywjTecKLdF6UULke6h1tAIsKBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d63480abebfebf0e3e5f2138565f4fdc54ecd434a3ab7fe91c8bfa49b233ef5","last_reissued_at":"2026-07-05T08:49:25.023572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:25.023572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Predictive Pipelined Decoding: A Compute-Latency Trade-off for Exact LLM Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dimitris Papailiopoulos, Gibbeum Lee, Jaewoong Cho, Kangwook Lee, Seongjun Yang","submitted_at":"2023-07-12T04:28:41Z","abstract_excerpt":"This paper presents \"Predictive Pipelined Decoding (PPD),\" an approach that speeds up greedy decoding in Large Language Models (LLMs) while maintaining the exact same output as the original decoding. Unlike conventional strategies, PPD employs additional compute resources to parallelize the initiation of subsequent token decoding during the current token decoding. This method reduces decoding latency and reshapes the understanding of trade-offs in LLM decoding strategies. We have developed a theoretical framework that allows us to analyze the trade-off between computation and latency. Using th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.05908","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.05908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.05908","created_at":"2026-07-05T08:49:25.023628+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.05908v2","created_at":"2026-07-05T08:49:25.023628+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.05908","created_at":"2026-07-05T08:49:25.023628+00:00"},{"alias_kind":"pith_short_12","alias_value":"RVRUQCV6X7V7","created_at":"2026-07-05T08:49:25.023628+00:00"},{"alias_kind":"pith_short_16","alias_value":"RVRUQCV6X7V7BY7F","created_at":"2026-07-05T08:49:25.023628+00:00"},{"alias_kind":"pith_short_8","alias_value":"RVRUQCV6","created_at":"2026-07-05T08:49:25.023628+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2401.15077","citing_title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X","json":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X.json","graph_json":"https://pith.science/api/pith-number/RVRUQCV6X7V7BY7F6IJYKZPU7X/graph.json","events_json":"https://pith.science/api/pith-number/RVRUQCV6X7V7BY7F6IJYKZPU7X/events.json","paper":"https://pith.science/paper/RVRUQCV6"},"agent_actions":{"view_html":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X","download_json":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X.json","view_paper":"https://pith.science/paper/RVRUQCV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.05908&json=true","fetch_graph":"https://pith.science/api/pith-number/RVRUQCV6X7V7BY7F6IJYKZPU7X/graph.json","fetch_events":"https://pith.science/api/pith-number/RVRUQCV6X7V7BY7F6IJYKZPU7X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X/action/storage_attestation","attest_author":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X/action/author_attestation","sign_citation":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X/action/citation_signature","submit_replication":"https://pith.science/pith/RVRUQCV6X7V7BY7F6IJYKZPU7X/action/replication_record"}},"created_at":"2026-07-05T08:49:25.023628+00:00","updated_at":"2026-07-05T08:49:25.023628+00:00"}