{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XZSBIJGQMWOOHGSQBRHWQV65IG","short_pith_number":"pith:XZSBIJGQ","schema_version":"1.0","canonical_sha256":"be641424d0659ce39a500c4f6857dd41a59d7fe0c4e5cc58e7a99b64c6121890","source":{"kind":"arxiv","id":"2311.04934","version":2},"attestation_state":"computed","paper":{"title":"Prompt Cache: Modular Attention Reuse for Low-Latency Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anurag Khandelwal, Guojun Chen, In Gim, Lin Zhong, Nikhil Sarda, Seung-seob Lee","submitted_at":"2023-11-07T18:17:05Z","abstract_excerpt":"We present Prompt Cache, an approach for accelerating inference for large language models (LLM) by reusing attention states across different LLM prompts. Many input prompts have overlapping text segments, such as system messages, prompt templates, and documents provided for context. Our key insight is that by precomputing and storing the attention states of these frequently occurring text segments on the inference server, we can efficiently reuse them when these segments appear in user prompts. Prompt Cache employs a schema to explicitly define such reusable text segments, called prompt module"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.04934","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-07T18:17:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9f3977bf5b278c6926d72643eace933bd7adba2afda577452fb7d892be7c8b4b","abstract_canon_sha256":"e560f0a88acc8c91f73090d22acc42e24b7114b519bc799fa960323edd94b50a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:11:55.376153Z","signature_b64":"xH9yTITGV00mLJX9NL1dSgwpu/6n8yfSkVuDOkxbQ4wnSHAHnRAUXYyMY2LeVa5GZGgssAYTdB2YvQOQ3jigBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be641424d0659ce39a500c4f6857dd41a59d7fe0c4e5cc58e7a99b64c6121890","last_reissued_at":"2026-07-05T08:11:55.375709Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:11:55.375709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prompt Cache: Modular Attention Reuse for Low-Latency Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anurag Khandelwal, Guojun Chen, In Gim, Lin Zhong, Nikhil Sarda, Seung-seob Lee","submitted_at":"2023-11-07T18:17:05Z","abstract_excerpt":"We present Prompt Cache, an approach for accelerating inference for large language models (LLM) by reusing attention states across different LLM prompts. Many input prompts have overlapping text segments, such as system messages, prompt templates, and documents provided for context. Our key insight is that by precomputing and storing the attention states of these frequently occurring text segments on the inference server, we can efficiently reuse them when these segments appear in user prompts. Prompt Cache employs a schema to explicitly define such reusable text segments, called prompt module"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.04934","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.04934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.04934","created_at":"2026-07-05T08:11:55.375769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.04934v2","created_at":"2026-07-05T08:11:55.375769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.04934","created_at":"2026-07-05T08:11:55.375769+00:00"},{"alias_kind":"pith_short_12","alias_value":"XZSBIJGQMWOO","created_at":"2026-07-05T08:11:55.375769+00:00"},{"alias_kind":"pith_short_16","alias_value":"XZSBIJGQMWOOHGSQ","created_at":"2026-07-05T08:11:55.375769+00:00"},{"alias_kind":"pith_short_8","alias_value":"XZSBIJGQ","created_at":"2026-07-05T08:11:55.375769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21676","citing_title":"CRAwLeR -- Cross-Reference Aware Legal Retrieval","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20537","citing_title":"Execution-State Capsules: Graph-Bound Execution-State Checkpoint and Restore for Low-Latency, Small-Batch, On-Device Physical-AI Serving","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28565","citing_title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09441","citing_title":"SIFT: Selective-Index For Fast Compute of RAG Prefill by Exploiting Attention Invariance","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01502","citing_title":"Move the Query, Not the Cache: Characterizing Cross-Instance Latent Attention Redistribution Across GPU Fabrics","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28565","citing_title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30613","citing_title":"CacheProbe: Auditing Prompt Cache Isolation in Gateway APIs","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13126","citing_title":"MiniPIC: Flexible Position-Independent Caching in <100LOC","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26924","citing_title":"A Deterministic Control Plane for LLM Coding Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10726","citing_title":"PrefixWall: Mitigating Prefix Caching Side Channels in Shared LLM Systems","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09725","citing_title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11232","citing_title":"Rethinking LLMOps for Fraud and AML: Building a Compliance-Grade LLM Serving Stack","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2312.07104","citing_title":"SGLang: Efficient Execution of Structured Language Model Programs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20021","citing_title":"Continuous Semantic Caching for Low-Cost LLM Serving","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16864","citing_title":"HieraSparse: Hierarchical Semi-Structured Sparse KV Attention","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG","json":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG.json","graph_json":"https://pith.science/api/pith-number/XZSBIJGQMWOOHGSQBRHWQV65IG/graph.json","events_json":"https://pith.science/api/pith-number/XZSBIJGQMWOOHGSQBRHWQV65IG/events.json","paper":"https://pith.science/paper/XZSBIJGQ"},"agent_actions":{"view_html":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG","download_json":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG.json","view_paper":"https://pith.science/paper/XZSBIJGQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.04934&json=true","fetch_graph":"https://pith.science/api/pith-number/XZSBIJGQMWOOHGSQBRHWQV65IG/graph.json","fetch_events":"https://pith.science/api/pith-number/XZSBIJGQMWOOHGSQBRHWQV65IG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG/action/storage_attestation","attest_author":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG/action/author_attestation","sign_citation":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG/action/citation_signature","submit_replication":"https://pith.science/pith/XZSBIJGQMWOOHGSQBRHWQV65IG/action/replication_record"}},"created_at":"2026-07-05T08:11:55.375769+00:00","updated_at":"2026-07-05T08:11:55.375769+00:00"}