{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AB5Q3VUB7MVNQQVPS7XDWABFRD","short_pith_number":"pith:AB5Q3VUB","schema_version":"1.0","canonical_sha256":"007b0dd681fb2ad842af97ee3b002588ed40f7e9b096c987d57f6b1061a99b0d","source":{"kind":"arxiv","id":"2503.16525","version":2},"attestation_state":"computed","paper":{"title":"KVShare: An LLM Service System with Efficient and Effective Multi-Tenant KV Cache Reuse","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Deyu Zhang, Huan Yang, Mingzhe Huang, Renji Zhang, Weijun Wang, Yin Tang, Yuanchun Li, Yunxin Liu","submitted_at":"2025-03-17T16:43:35Z","abstract_excerpt":"Recent advances in long-text understanding have pushed the context length of large language models (LLMs) up to one million tokens. It boosts LLMs's accuracy and reasoning capacity but causes exorbitant computational costs and unsatisfactory Time to First Token (TTFT). KV cache reuse, which reuses the exact same KV cache of prefixes and templates or shares similar ones but with extra selective recomputation, offers a promising way to tackle this issue. However, prior studies overlook the cross-request KV reuse and the attention deviations introduced by new tokens during the decoding stage. In "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.16525","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-03-17T16:43:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ba5647dd7d1a06b689ea2bc1edc853eae119fc851ae635a72928312cea9974e5","abstract_canon_sha256":"f9cf4085c43b640861fcc56f5bdd965137edebaebcb0f37b33c38b5b8e05ec1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:58.389018Z","signature_b64":"LsyKhiThKGvC26a5Arh5n5yWcwVSGm9U3bNwmfhCgLJ+nUtmntHZsRpwgn63NJcDX0+fX5e8lp1mfg+vNtz2Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"007b0dd681fb2ad842af97ee3b002588ed40f7e9b096c987d57f6b1061a99b0d","last_reissued_at":"2026-07-05T11:03:58.388393Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:58.388393Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KVShare: An LLM Service System with Efficient and Effective Multi-Tenant KV Cache Reuse","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Deyu Zhang, Huan Yang, Mingzhe Huang, Renji Zhang, Weijun Wang, Yin Tang, Yuanchun Li, Yunxin Liu","submitted_at":"2025-03-17T16:43:35Z","abstract_excerpt":"Recent advances in long-text understanding have pushed the context length of large language models (LLMs) up to one million tokens. It boosts LLMs's accuracy and reasoning capacity but causes exorbitant computational costs and unsatisfactory Time to First Token (TTFT). KV cache reuse, which reuses the exact same KV cache of prefixes and templates or shares similar ones but with extra selective recomputation, offers a promising way to tackle this issue. However, prior studies overlook the cross-request KV reuse and the attention deviations introduced by new tokens during the decoding stage. In "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.16525","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.16525/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.16525","created_at":"2026-07-05T11:03:58.388486+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.16525v2","created_at":"2026-07-05T11:03:58.388486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.16525","created_at":"2026-07-05T11:03:58.388486+00:00"},{"alias_kind":"pith_short_12","alias_value":"AB5Q3VUB7MVN","created_at":"2026-07-05T11:03:58.388486+00:00"},{"alias_kind":"pith_short_16","alias_value":"AB5Q3VUB7MVNQQVP","created_at":"2026-07-05T11:03:58.388486+00:00"},{"alias_kind":"pith_short_8","alias_value":"AB5Q3VUB","created_at":"2026-07-05T11:03:58.388486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01299","citing_title":"HYPIC: Accelerating Hybrid-Attention LLM Serving with Position-Independent Caching","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01751","citing_title":"SparseX: Efficient Segment-Level KV Cache Sharing for Interleaved LLM Serving","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01065","citing_title":"Leyline: KV Cache Directives for Agentic Inference","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14371","citing_title":"OxyGen: Unified KV Cache Management for VLA Inference under Multi-Task Parallelism","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08075","citing_title":"Dual-Pool Token-Budget Routing for Cost-Efficient and Reliable LLM Serving","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD","json":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD.json","graph_json":"https://pith.science/api/pith-number/AB5Q3VUB7MVNQQVPS7XDWABFRD/graph.json","events_json":"https://pith.science/api/pith-number/AB5Q3VUB7MVNQQVPS7XDWABFRD/events.json","paper":"https://pith.science/paper/AB5Q3VUB"},"agent_actions":{"view_html":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD","download_json":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD.json","view_paper":"https://pith.science/paper/AB5Q3VUB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.16525&json=true","fetch_graph":"https://pith.science/api/pith-number/AB5Q3VUB7MVNQQVPS7XDWABFRD/graph.json","fetch_events":"https://pith.science/api/pith-number/AB5Q3VUB7MVNQQVPS7XDWABFRD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD/action/storage_attestation","attest_author":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD/action/author_attestation","sign_citation":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD/action/citation_signature","submit_replication":"https://pith.science/pith/AB5Q3VUB7MVNQQVPS7XDWABFRD/action/replication_record"}},"created_at":"2026-07-05T11:03:58.388486+00:00","updated_at":"2026-07-05T11:03:58.388486+00:00"}