{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EMSCH3ASNDFOZ56TCOJLC7E3PF","short_pith_number":"pith:EMSCH3AS","schema_version":"1.0","canonical_sha256":"232423ec1268caecf7d31392b17c9b794330799c6fd8fa25c94dfd4d7bc618b9","source":{"kind":"arxiv","id":"2404.11912","version":3},"attestation_state":"computed","paper":{"title":"TriForce: Lossless Acceleration of Long Sequence Generation with Hierarchical Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Hanshi Sun, Xinyu Yang, Yuandong Tian, Zhuoming Chen","submitted_at":"2024-04-18T05:25:54Z","abstract_excerpt":"With large language models (LLMs) widely deployed in long content generation recently, there has emerged an increasing demand for efficient long-sequence inference support. However, key-value (KV) cache, which is stored to avoid re-computation, has emerged as a critical bottleneck by growing linearly in size with the sequence length. Due to the auto-regressive nature of LLMs, the entire KV cache will be loaded for every generated token, resulting in low utilization of computational cores and high latency. While various compression methods for KV cache have been proposed to alleviate this issue"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.11912","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-18T05:25:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"378718ec933c48217958b0291c823fb49d822f7f60cb0e62adfc3dda66e36896","abstract_canon_sha256":"97436673826914a3f9bd595f2f314f6dc2aaf080ee843e47b6d18a6a89559245"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:50.795606Z","signature_b64":"JnpW6zRXaXrsMEQ7IidkIe96E3B/ZwCvzr198nRaPZ4XARtQM9GS67OuT4x552463f6Ed207hVMOdb7REH/pCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"232423ec1268caecf7d31392b17c9b794330799c6fd8fa25c94dfd4d7bc618b9","last_reissued_at":"2026-07-05T08:51:50.795142Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:50.795142Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TriForce: Lossless Acceleration of Long Sequence Generation with Hierarchical Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Hanshi Sun, Xinyu Yang, Yuandong Tian, Zhuoming Chen","submitted_at":"2024-04-18T05:25:54Z","abstract_excerpt":"With large language models (LLMs) widely deployed in long content generation recently, there has emerged an increasing demand for efficient long-sequence inference support. However, key-value (KV) cache, which is stored to avoid re-computation, has emerged as a critical bottleneck by growing linearly in size with the sequence length. Due to the auto-regressive nature of LLMs, the entire KV cache will be loaded for every generated token, resulting in low utilization of computational cores and high latency. While various compression methods for KV cache have been proposed to alleviate this issue"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.11912","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.11912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.11912","created_at":"2026-07-05T08:51:50.795205+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.11912v3","created_at":"2026-07-05T08:51:50.795205+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.11912","created_at":"2026-07-05T08:51:50.795205+00:00"},{"alias_kind":"pith_short_12","alias_value":"EMSCH3ASNDFO","created_at":"2026-07-05T08:51:50.795205+00:00"},{"alias_kind":"pith_short_16","alias_value":"EMSCH3ASNDFOZ56T","created_at":"2026-07-05T08:51:50.795205+00:00"},{"alias_kind":"pith_short_8","alias_value":"EMSCH3AS","created_at":"2026-07-05T08:51:50.795205+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24957","citing_title":"Dustin: Draft-Augmented Sparse Verification for Efficient Long-Context Generation with Speculative Decoding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29563","citing_title":"Coverage-Driven KV Cache Eviction for Efficient and Improved Inference of LLM","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28678","citing_title":"DREAM-R: Multimodal Speculative Reasoning with RL-Based Refined Drafting, Precise Verification, and Fully Parallel Execution","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00144","citing_title":"BudgetDraft: Acceptance-Aware Multi-View Training for Sparse-KV Speculative Decoding","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00535","citing_title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20104","citing_title":"Draft Less, Retrieve More: Hybrid Tree Construction for Speculative Decoding","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2407.11550","citing_title":"Ada-KV: Optimizing KV Cache Eviction by Adaptive Budget Allocation for Efficient LLM Inference","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF","json":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF.json","graph_json":"https://pith.science/api/pith-number/EMSCH3ASNDFOZ56TCOJLC7E3PF/graph.json","events_json":"https://pith.science/api/pith-number/EMSCH3ASNDFOZ56TCOJLC7E3PF/events.json","paper":"https://pith.science/paper/EMSCH3AS"},"agent_actions":{"view_html":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF","download_json":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF.json","view_paper":"https://pith.science/paper/EMSCH3AS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.11912&json=true","fetch_graph":"https://pith.science/api/pith-number/EMSCH3ASNDFOZ56TCOJLC7E3PF/graph.json","fetch_events":"https://pith.science/api/pith-number/EMSCH3ASNDFOZ56TCOJLC7E3PF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF/action/storage_attestation","attest_author":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF/action/author_attestation","sign_citation":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF/action/citation_signature","submit_replication":"https://pith.science/pith/EMSCH3ASNDFOZ56TCOJLC7E3PF/action/replication_record"}},"created_at":"2026-07-05T08:51:50.795205+00:00","updated_at":"2026-07-05T08:51:50.795205+00:00"}