{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QY6KVBYS2ZM2J5OWIR6WJHEVTH","short_pith_number":"pith:QY6KVBYS","schema_version":"1.0","canonical_sha256":"863caa8712d659a4f5d6447d649c9599d067e4a07a2d60abfaec6255a0384676","source":{"kind":"arxiv","id":"2411.19379","version":3},"attestation_state":"computed","paper":{"title":"Marconi: Prefix Caching for the Era of Hybrid LLMs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Can Karakus, Luca Zancato, Ravi Netravali, Rui Pan, Tri Dao, Yida Wang, Zhen Jia, Zhuang Wang","submitted_at":"2024-11-28T21:10:20Z","abstract_excerpt":"Hybrid models that combine the language modeling capabilities of Attention layers with the efficiency of Recurrent layers (e.g., State Space Models) have gained traction in practically supporting long contexts in Large Language Model serving. Yet, the unique properties of these models complicate the usage of complementary efficiency optimizations such as prefix caching that skip redundant computations across requests. Most notably, their use of in-place state updates for recurrent layers precludes rolling back cache entries for partial sequence overlaps, and instead mandates only exact-match c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.19379","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.DC","submitted_at":"2024-11-28T21:10:20Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ed40a956f625872be41c909dd60e4bc3cb2f74ae5138a19ea2b8b0e0788d7d92","abstract_canon_sha256":"7436ffa074e70a00e3f98fac4ccf3641091d968cd18a2a533d89b9e78d29a7ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:59.003219Z","signature_b64":"T3hxYFbjIhC3zXL8JhXEaFQVNgq/eSRdhFaiOo5rXGE7EmLWBkZQGg8goKdaH4pGT81aIF3rmWk64N9Mbl4wCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"863caa8712d659a4f5d6447d649c9599d067e4a07a2d60abfaec6255a0384676","last_reissued_at":"2026-07-05T10:46:59.002711Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:59.002711Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Marconi: Prefix Caching for the Era of Hybrid LLMs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Can Karakus, Luca Zancato, Ravi Netravali, Rui Pan, Tri Dao, Yida Wang, Zhen Jia, Zhuang Wang","submitted_at":"2024-11-28T21:10:20Z","abstract_excerpt":"Hybrid models that combine the language modeling capabilities of Attention layers with the efficiency of Recurrent layers (e.g., State Space Models) have gained traction in practically supporting long contexts in Large Language Model serving. Yet, the unique properties of these models complicate the usage of complementary efficiency optimizations such as prefix caching that skip redundant computations across requests. Most notably, their use of in-place state updates for recurrent layers precludes rolling back cache entries for partial sequence overlaps, and instead mandates only exact-match c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.19379","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.19379/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.19379","created_at":"2026-07-05T10:46:59.002778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.19379v3","created_at":"2026-07-05T10:46:59.002778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.19379","created_at":"2026-07-05T10:46:59.002778+00:00"},{"alias_kind":"pith_short_12","alias_value":"QY6KVBYS2ZM2","created_at":"2026-07-05T10:46:59.002778+00:00"},{"alias_kind":"pith_short_16","alias_value":"QY6KVBYS2ZM2J5OW","created_at":"2026-07-05T10:46:59.002778+00:00"},{"alias_kind":"pith_short_8","alias_value":"QY6KVBYS","created_at":"2026-07-05T10:46:59.002778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21238","citing_title":"Recency/Frequency Adaptive KV Caching for Large Language Model Serving","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21016","citing_title":"Gated KalmaNet: A Fading Memory Layer Through Test-Time Ridge Regression","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18825","citing_title":"Not All Tokens Are Worth Caching: Learning Semantic-Aware Eviction for LLM Prefix Caches","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2512.16056","citing_title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05696","citing_title":"Irminsul: MLA-Native Position-Independent Caching for Agentic LLM Serving","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06068","citing_title":"VibeServe: Can AI Agents Build Bespoke LLM Serving Systems?","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05219","citing_title":"Sparse Prefix Caching for Hybrid and Recurrent LLM Serving","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH","json":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH.json","graph_json":"https://pith.science/api/pith-number/QY6KVBYS2ZM2J5OWIR6WJHEVTH/graph.json","events_json":"https://pith.science/api/pith-number/QY6KVBYS2ZM2J5OWIR6WJHEVTH/events.json","paper":"https://pith.science/paper/QY6KVBYS"},"agent_actions":{"view_html":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH","download_json":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH.json","view_paper":"https://pith.science/paper/QY6KVBYS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.19379&json=true","fetch_graph":"https://pith.science/api/pith-number/QY6KVBYS2ZM2J5OWIR6WJHEVTH/graph.json","fetch_events":"https://pith.science/api/pith-number/QY6KVBYS2ZM2J5OWIR6WJHEVTH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH/action/storage_attestation","attest_author":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH/action/author_attestation","sign_citation":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH/action/citation_signature","submit_replication":"https://pith.science/pith/QY6KVBYS2ZM2J5OWIR6WJHEVTH/action/replication_record"}},"created_at":"2026-07-05T10:46:59.002778+00:00","updated_at":"2026-07-05T10:46:59.002778+00:00"}