{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SDQOVIB33E2SRE2SFQZZNL6XOI","short_pith_number":"pith:SDQOVIB3","schema_version":"1.0","canonical_sha256":"90e0eaa03bd9352893522c3396afd7723acb3d2416c48d6a3c43e8965d6abe0d","source":{"kind":"arxiv","id":"2412.15605","version":2},"attestation_state":"computed","paper":{"title":"Don't Do RAG: When Cache-Augmented Generation is All You Need for Knowledge Tasks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Brian J Chan, Chao-Ting Chen, Hen-Hsen Huang, Jui-Hung Cheng","submitted_at":"2024-12-20T06:58:32Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has gained traction as a powerful approach for enhancing language models by integrating external knowledge sources. However, RAG introduces challenges such as retrieval latency, potential errors in document selection, and increased system complexity. With the advent of large language models (LLMs) featuring significantly extended context windows, this paper proposes an alternative paradigm, cache-augmented generation (CAG) that bypasses real-time retrieval. Our method involves preloading all relevant resources, especially when the documents or knowledge for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15605","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-20T06:58:32Z","cross_cats_sorted":[],"title_canon_sha256":"b1ffc2620e9bda8438366d34b01a0ff22c06f6ab96d05cc52928297b84a451a0","abstract_canon_sha256":"a3137e231af7ef26948185cb7f02d0b9b0970227eac3291574d38705affb2543"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:47.328041Z","signature_b64":"soxER35wep6BvT4gJGTHAslCc0ZjwbJTB6OGA4KXFiAnbgozWiSlBe+yX/BDjJdW2JJ2QpJ34UUdhjhIeuc4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90e0eaa03bd9352893522c3396afd7723acb3d2416c48d6a3c43e8965d6abe0d","last_reissued_at":"2026-07-05T10:18:47.327465Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:47.327465Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't Do RAG: When Cache-Augmented Generation is All You Need for Knowledge Tasks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Brian J Chan, Chao-Ting Chen, Hen-Hsen Huang, Jui-Hung Cheng","submitted_at":"2024-12-20T06:58:32Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has gained traction as a powerful approach for enhancing language models by integrating external knowledge sources. However, RAG introduces challenges such as retrieval latency, potential errors in document selection, and increased system complexity. With the advent of large language models (LLMs) featuring significantly extended context windows, this paper proposes an alternative paradigm, cache-augmented generation (CAG) that bypasses real-time retrieval. Our method involves preloading all relevant resources, especially when the documents or knowledge for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15605","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15605/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15605","created_at":"2026-07-05T10:18:47.327531+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15605v2","created_at":"2026-07-05T10:18:47.327531+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15605","created_at":"2026-07-05T10:18:47.327531+00:00"},{"alias_kind":"pith_short_12","alias_value":"SDQOVIB33E2S","created_at":"2026-07-05T10:18:47.327531+00:00"},{"alias_kind":"pith_short_16","alias_value":"SDQOVIB33E2SRE2S","created_at":"2026-07-05T10:18:47.327531+00:00"},{"alias_kind":"pith_short_8","alias_value":"SDQOVIB3","created_at":"2026-07-05T10:18:47.327531+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26105","citing_title":"Context Recycling for Long-Horizon LLM Inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27494","citing_title":"Grounded Cache Routing for Retrieval-Augmented Generation: When Is It Safe to Reuse an Answer?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07068","citing_title":"WiCER: Wiki-memory Compile, Evaluate, Refine Iterative Knowledge Compilation for LLM Wiki Systems","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI","json":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI.json","graph_json":"https://pith.science/api/pith-number/SDQOVIB33E2SRE2SFQZZNL6XOI/graph.json","events_json":"https://pith.science/api/pith-number/SDQOVIB33E2SRE2SFQZZNL6XOI/events.json","paper":"https://pith.science/paper/SDQOVIB3"},"agent_actions":{"view_html":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI","download_json":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI.json","view_paper":"https://pith.science/paper/SDQOVIB3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15605&json=true","fetch_graph":"https://pith.science/api/pith-number/SDQOVIB33E2SRE2SFQZZNL6XOI/graph.json","fetch_events":"https://pith.science/api/pith-number/SDQOVIB33E2SRE2SFQZZNL6XOI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI/action/storage_attestation","attest_author":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI/action/author_attestation","sign_citation":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI/action/citation_signature","submit_replication":"https://pith.science/pith/SDQOVIB33E2SRE2SFQZZNL6XOI/action/replication_record"}},"created_at":"2026-07-05T10:18:47.327531+00:00","updated_at":"2026-07-05T10:18:47.327531+00:00"}