{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:23GPZM4GKXULDKNNFGY2JD3XB3","short_pith_number":"pith:23GPZM4G","schema_version":"1.0","canonical_sha256":"d6ccfcb38655e8b1a9ad29b1a48f770eec4b9c79a1dc8d274f74940966c59b45","source":{"kind":"arxiv","id":"2506.06266","version":3},"attestation_state":"computed","paper":{"title":"Cartridges: Lightweight and general-purpose long context representations via self-study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Atri Rudra, Azalia Mirhoseini, Christopher Re, Dylan Zinsley, Emily Liu, James Zou, Neel Guha, Ryan Ehrlich, Sabri Eyuboglu, Simran Arora, Will Tennien","submitted_at":"2025-06-06T17:48:23Z","abstract_excerpt":"Large language models are often used to answer queries grounded in large text corpora (e.g. codebases, legal documents, or chat histories) by placing the entire corpus in the context window and leveraging in-context learning (ICL). Although current models support contexts of 100K-1M tokens, this setup is costly to serve because the memory consumption of the KV cache scales with input length. We explore an alternative: training a smaller KV cache offline on each corpus. At inference time, we load this trained KV cache, which we call a Cartridge, and decode a response. Critically, the cost of tr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.06266","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-06T17:48:23Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"c84e06e14bf5bfd550ea6f1c5387beea382a8424fcc5d75523afa2e5a78b1df3","abstract_canon_sha256":"f01cff37511396def0558cf705175b6452c9657b38737accaeea28ef7beb9f7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:55.413624Z","signature_b64":"ptB1wBPUkc0j5EZcXFROXLEbHSg5Tk+WldqKX4XuEhCSGKNAngV7T+EWkxSil6igjVNSKXHEz2hz/l5exUJcBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d6ccfcb38655e8b1a9ad29b1a48f770eec4b9c79a1dc8d274f74940966c59b45","last_reissued_at":"2026-07-05T11:20:55.413097Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:55.413097Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cartridges: Lightweight and general-purpose long context representations via self-study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Atri Rudra, Azalia Mirhoseini, Christopher Re, Dylan Zinsley, Emily Liu, James Zou, Neel Guha, Ryan Ehrlich, Sabri Eyuboglu, Simran Arora, Will Tennien","submitted_at":"2025-06-06T17:48:23Z","abstract_excerpt":"Large language models are often used to answer queries grounded in large text corpora (e.g. codebases, legal documents, or chat histories) by placing the entire corpus in the context window and leveraging in-context learning (ICL). Although current models support contexts of 100K-1M tokens, this setup is costly to serve because the memory consumption of the KV cache scales with input length. We explore an alternative: training a smaller KV cache offline on each corpus. At inference time, we load this trained KV cache, which we call a Cartridge, and decode a response. Critically, the cost of tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.06266","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.06266/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.06266","created_at":"2026-07-05T11:20:55.413163+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.06266v3","created_at":"2026-07-05T11:20:55.413163+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.06266","created_at":"2026-07-05T11:20:55.413163+00:00"},{"alias_kind":"pith_short_12","alias_value":"23GPZM4GKXUL","created_at":"2026-07-05T11:20:55.413163+00:00"},{"alias_kind":"pith_short_16","alias_value":"23GPZM4GKXULDKNN","created_at":"2026-07-05T11:20:55.413163+00:00"},{"alias_kind":"pith_short_8","alias_value":"23GPZM4G","created_at":"2026-07-05T11:20:55.413163+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07847","citing_title":"When Does Continual Learning Require Learning","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12400","citing_title":"Doc-to-Atom: Learning to Compile and Compose Memory Atoms","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09659","citing_title":"End-to-End Context Compression at Scale","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07878","citing_title":"Still: Amortized KV Cache Compaction in a Single Forward Pass","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06154","citing_title":"Amortizing Federated Adaptation: Hypernetwork Driven LoRA for Personalized Foundation Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05661","citing_title":"Continual Learning Bench: Evaluating Frontier AI Systems in Real-World Stateful Environments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03979","citing_title":"Language Models Need Sleep: Learning to Self-Modify and Consolidate Memories","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32002","citing_title":"Self-Study Reconsidered: The Hidden Fragility of Learning from Self-Generated QA","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29503","citing_title":"The Verbose Context Problem in Medical Records","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26099","citing_title":"Do Language Models Need Sleep? Offline Recurrence for Improved Online Inference","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28889","citing_title":"Context Distillation as Latent Memory Management","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29075","citing_title":"Knowledge Offloading: Decomposing LLMs into Sparse Backbones and Memory Modules","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23296","citing_title":"Parallel Context Compaction for Long-Horizon LLM Agent Serving","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03092","citing_title":"SnapStream: Efficient Long Sequence Decoding on Dataflow Accelerators","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07957","citing_title":"MIRIX: Multi-Agent Memory System for LLM-Based Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14217","citing_title":"PreFT: Prefill-only finetuning for efficient inference","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11628","citing_title":"Back to Basics: Let Conversational Agents Remember with Just Retrieval and Generation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3","json":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3.json","graph_json":"https://pith.science/api/pith-number/23GPZM4GKXULDKNNFGY2JD3XB3/graph.json","events_json":"https://pith.science/api/pith-number/23GPZM4GKXULDKNNFGY2JD3XB3/events.json","paper":"https://pith.science/paper/23GPZM4G"},"agent_actions":{"view_html":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3","download_json":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3.json","view_paper":"https://pith.science/paper/23GPZM4G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.06266&json=true","fetch_graph":"https://pith.science/api/pith-number/23GPZM4GKXULDKNNFGY2JD3XB3/graph.json","fetch_events":"https://pith.science/api/pith-number/23GPZM4GKXULDKNNFGY2JD3XB3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3/action/storage_attestation","attest_author":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3/action/author_attestation","sign_citation":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3/action/citation_signature","submit_replication":"https://pith.science/pith/23GPZM4GKXULDKNNFGY2JD3XB3/action/replication_record"}},"created_at":"2026-07-05T11:20:55.413163+00:00","updated_at":"2026-07-05T11:20:55.413163+00:00"}