{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JRNDNKXZEBP7IA6TOJJH5CLUEW","short_pith_number":"pith:JRNDNKXZ","schema_version":"1.0","canonical_sha256":"4c5a36aaf9205ff403d372527e89742588c90699c74567d00878828287f9961d","source":{"kind":"arxiv","id":"2404.12457","version":2},"attestation_state":"computed","paper":{"title":"RAGCache: Efficient Knowledge Caching for Retrieval-Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.DC","authors_text":"Chao Jin, Fangyue Liu, Xin Jin, Xin Liu, Xuanlin Jiang, Xuanzhe Liu, Zili Zhang","submitted_at":"2024-04-18T18:32:30Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) has shown significant improvements in various natural language processing tasks by integrating the strengths of large language models (LLMs) and external knowledge databases. However, RAG introduces long sequence generation and leads to high computation and memory costs. We propose RAGCache, a novel multilevel dynamic caching system tailored for RAG. Our analysis benchmarks current RAG systems, pinpointing the performance bottleneck (i.e., long sequence due to knowledge injection) and optimization opportunities (i.e., caching knowledge's intermediate states"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.12457","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-04-18T18:32:30Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"51026f29864be3161bf0fa1753649e51679a9cc95ce91ae80963e0cf626a873b","abstract_canon_sha256":"9c102747ed5f8d4c5b523d578c5532e0795f162cfd6dff95c09a61bd16fca664"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:11:59.362679Z","signature_b64":"yd2wT6yshjZJDLANZIEi00layeO6qgEhqostnQi/bz8FDZA0pxFvbu8GMmIObXZFHf8WOa0Lz95aEFahY8YwDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c5a36aaf9205ff403d372527e89742588c90699c74567d00878828287f9961d","last_reissued_at":"2026-07-05T08:11:59.362158Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:11:59.362158Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RAGCache: Efficient Knowledge Caching for Retrieval-Augmented Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.DC","authors_text":"Chao Jin, Fangyue Liu, Xin Jin, Xin Liu, Xuanlin Jiang, Xuanzhe Liu, Zili Zhang","submitted_at":"2024-04-18T18:32:30Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) has shown significant improvements in various natural language processing tasks by integrating the strengths of large language models (LLMs) and external knowledge databases. However, RAG introduces long sequence generation and leads to high computation and memory costs. We propose RAGCache, a novel multilevel dynamic caching system tailored for RAG. Our analysis benchmarks current RAG systems, pinpointing the performance bottleneck (i.e., long sequence due to knowledge injection) and optimization opportunities (i.e., caching knowledge's intermediate states"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12457","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.12457","created_at":"2026-07-05T08:11:59.362215+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.12457v2","created_at":"2026-07-05T08:11:59.362215+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12457","created_at":"2026-07-05T08:11:59.362215+00:00"},{"alias_kind":"pith_short_12","alias_value":"JRNDNKXZEBP7","created_at":"2026-07-05T08:11:59.362215+00:00"},{"alias_kind":"pith_short_16","alias_value":"JRNDNKXZEBP7IA6T","created_at":"2026-07-05T08:11:59.362215+00:00"},{"alias_kind":"pith_short_8","alias_value":"JRNDNKXZ","created_at":"2026-07-05T08:11:59.362215+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17107","citing_title":"Models Take Notes at Prefill: KV Cache Can Be Editable and Composable","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09441","citing_title":"SIFT: Selective-Index For Fast Compute of RAG Prefill by Exploiting Attention Invariance","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27494","citing_title":"Grounded Cache Routing for Retrieval-Augmented Generation: When Is It Safe to Reuse an Answer?","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04302","citing_title":"LazyAttention: Efficient Retrieval-Augmented Generation with Deferred Positional Encoding","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13097","citing_title":"Functional Cache Grafting: Robust and Rapid Code-Policy Synthesis for Embodied Agents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2407.13193","citing_title":"Retrieval-Augmented Generation for Natural Language Processing: A Survey","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03594","citing_title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10129","citing_title":"CacheClip: Accelerating RAG with Effective KV Cache Reuse","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20630","citing_title":"Evaluating Temporal Semantic Caching and Workflow Optimization in Agentic Plan-Execute Pipelines","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17934","citing_title":"AtlasKV: Augmenting LLMs with Billion-Scale Knowledge Graphs in 20GB VRAM","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09725","citing_title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW","json":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW.json","graph_json":"https://pith.science/api/pith-number/JRNDNKXZEBP7IA6TOJJH5CLUEW/graph.json","events_json":"https://pith.science/api/pith-number/JRNDNKXZEBP7IA6TOJJH5CLUEW/events.json","paper":"https://pith.science/paper/JRNDNKXZ"},"agent_actions":{"view_html":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW","download_json":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW.json","view_paper":"https://pith.science/paper/JRNDNKXZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.12457&json=true","fetch_graph":"https://pith.science/api/pith-number/JRNDNKXZEBP7IA6TOJJH5CLUEW/graph.json","fetch_events":"https://pith.science/api/pith-number/JRNDNKXZEBP7IA6TOJJH5CLUEW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW/action/storage_attestation","attest_author":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW/action/author_attestation","sign_citation":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW/action/citation_signature","submit_replication":"https://pith.science/pith/JRNDNKXZEBP7IA6TOJJH5CLUEW/action/replication_record"}},"created_at":"2026-07-05T08:11:59.362215+00:00","updated_at":"2026-07-05T08:11:59.362215+00:00"}