{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YSVOIA6ECIWNF65635HLNQLLL5","short_pith_number":"pith:YSVOIA6E","schema_version":"1.0","canonical_sha256":"c4aae403c4122cd2fbbedf4eb6c16b5f4e076237632e24641aea6a1da7c7e836","source":{"kind":"arxiv","id":"2410.15332","version":3},"attestation_state":"computed","paper":{"title":"EPIC: Efficient Position-Independent Caching for Serving Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.DC","cs.PF"],"primary_cat":"cs.LG","authors_text":"Hao Feng, Haoyi Wang, Junhao Hu, Qin Zhang, Tao Xie, Tiancheng Hu, Weidong Wang, Wenrui Huang, Xusheng Chen, Yizhou Shan","submitted_at":"2024-10-20T08:42:29Z","abstract_excerpt":"Large Language Models (LLMs) show great capabilities in a wide range of applications, but serving them efficiently becomes increasingly challenging as requests (prompts) become more complex. Context caching improves serving performance by reusing Key-Value (KV) vectors, the intermediate representations of tokens that are repeated across requests. However, existing context caching requires exact prefix matches across requests, limiting reuse cases in settings such as few-shot learning and retrieval-augmented generation, where immutable content (e.g., documents) remains unchanged across requests"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15332","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-20T08:42:29Z","cross_cats_sorted":["cs.CL","cs.DC","cs.PF"],"title_canon_sha256":"3a26ef2947faf2813cc2e65210aa53fa60a98b79d832f714fbaedd3715583d28","abstract_canon_sha256":"f1e005b6a70b5683b231ac615ceab59785ae5b214bf417b8079c538c32566f8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:01.860865Z","signature_b64":"iz2A7gdkmw46ygd2u3Ojjb4jqtFhlrDjjoSXYaOkWui083Hn4SRgmnXUa4Ynh5XqdSls6cqiaaZ6C9xMOQPWBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4aae403c4122cd2fbbedf4eb6c16b5f4e076237632e24641aea6a1da7c7e836","last_reissued_at":"2026-07-05T11:10:01.860349Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:01.860349Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EPIC: Efficient Position-Independent Caching for Serving Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.DC","cs.PF"],"primary_cat":"cs.LG","authors_text":"Hao Feng, Haoyi Wang, Junhao Hu, Qin Zhang, Tao Xie, Tiancheng Hu, Weidong Wang, Wenrui Huang, Xusheng Chen, Yizhou Shan","submitted_at":"2024-10-20T08:42:29Z","abstract_excerpt":"Large Language Models (LLMs) show great capabilities in a wide range of applications, but serving them efficiently becomes increasingly challenging as requests (prompts) become more complex. Context caching improves serving performance by reusing Key-Value (KV) vectors, the intermediate representations of tokens that are repeated across requests. However, existing context caching requires exact prefix matches across requests, limiting reuse cases in settings such as few-shot learning and retrieval-augmented generation, where immutable content (e.g., documents) remains unchanged across requests"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15332","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15332/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15332","created_at":"2026-07-05T11:10:01.860413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15332v3","created_at":"2026-07-05T11:10:01.860413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15332","created_at":"2026-07-05T11:10:01.860413+00:00"},{"alias_kind":"pith_short_12","alias_value":"YSVOIA6ECIWN","created_at":"2026-07-05T11:10:01.860413+00:00"},{"alias_kind":"pith_short_16","alias_value":"YSVOIA6ECIWNF656","created_at":"2026-07-05T11:10:01.860413+00:00"},{"alias_kind":"pith_short_8","alias_value":"YSVOIA6E","created_at":"2026-07-05T11:10:01.860413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09441","citing_title":"SIFT: Selective-Index For Fast Compute of RAG Prefill by Exploiting Attention Invariance","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05875","citing_title":"QCFuse: Query-Aware Cache Fusion via Compressed View for Efficient RAG Serving","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27494","citing_title":"Grounded Cache Routing for Retrieval-Augmented Generation: When Is It Safe to Reuse an Answer?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13126","citing_title":"MiniPIC: Flexible Position-Independent Caching in <100LOC","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13097","citing_title":"Functional Cache Grafting: Robust and Rapid Code-Policy Synthesis for Embodied Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23640","citing_title":"CachePrune: Privacy-Aware and Fine-Grained KV Cache Sharing for Efficient LLM Inference","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03143","citing_title":"TokenDance: Scaling Multi-Agent LLM Serving via Collective KV Cache Sharing","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5","json":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5.json","graph_json":"https://pith.science/api/pith-number/YSVOIA6ECIWNF65635HLNQLLL5/graph.json","events_json":"https://pith.science/api/pith-number/YSVOIA6ECIWNF65635HLNQLLL5/events.json","paper":"https://pith.science/paper/YSVOIA6E"},"agent_actions":{"view_html":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5","download_json":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5.json","view_paper":"https://pith.science/paper/YSVOIA6E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15332&json=true","fetch_graph":"https://pith.science/api/pith-number/YSVOIA6ECIWNF65635HLNQLLL5/graph.json","fetch_events":"https://pith.science/api/pith-number/YSVOIA6ECIWNF65635HLNQLLL5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5/action/storage_attestation","attest_author":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5/action/author_attestation","sign_citation":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5/action/citation_signature","submit_replication":"https://pith.science/pith/YSVOIA6ECIWNF65635HLNQLLL5/action/replication_record"}},"created_at":"2026-07-05T11:10:01.860413+00:00","updated_at":"2026-07-05T11:10:01.860413+00:00"}