{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HFW7PH7UXOGHQ3UJDJDGJPABTZ","short_pith_number":"pith:HFW7PH7U","schema_version":"1.0","canonical_sha256":"396df79ff4bb8c786e891a4664bc019e66736e0c9e4bd7a0c0f6fb3f4cbaca5b","source":{"kind":"arxiv","id":"2405.04437","version":3},"attestation_state":"computed","paper":{"title":"vAttention: Dynamic Memory Management for Serving LLMs without PagedAttention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.OS"],"primary_cat":"cs.LG","authors_text":"Ajay Nayak, Ashish Panwar, Jayashree Mohan, Ramachandran Ramjee, Ramya Prabhu","submitted_at":"2024-05-07T16:00:32Z","abstract_excerpt":"PagedAttention is a popular approach for dynamic memory allocation in LLM serving systems. It enables on-demand allocation of GPU memory to mitigate KV cache fragmentation -- a phenomenon that crippled the batch size (and consequently throughput) in prior systems. However, in trying to allocate physical memory at runtime, PagedAttention ends up changing the virtual memory layout of the KV cache from contiguous to non-contiguous. Such a design leads to non-trivial programming and performance overheads.\n  We present vAttention -- an approach that mitigates fragmentation in physical memory while "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.04437","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-07T16:00:32Z","cross_cats_sorted":["cs.OS"],"title_canon_sha256":"be1ae0f2b5fa73be1a9071ab4219f99538b9ffadeccb3e81a542033330af91e4","abstract_canon_sha256":"21b700cc8e17e6b1db74b178c0ef756b4cd3338a50a2391b47e74bbdcead446c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:40.460958Z","signature_b64":"lioAWo/VWGj7iQPnmZen+3utV2AcbQpCskPEn6Ye79fhUek4PDJjVvVJDCem/OCqhuQsX679Y1P6D2MYiUaUAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"396df79ff4bb8c786e891a4664bc019e66736e0c9e4bd7a0c0f6fb3f4cbaca5b","last_reissued_at":"2026-07-05T10:06:40.460465Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:40.460465Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"vAttention: Dynamic Memory Management for Serving LLMs without PagedAttention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.OS"],"primary_cat":"cs.LG","authors_text":"Ajay Nayak, Ashish Panwar, Jayashree Mohan, Ramachandran Ramjee, Ramya Prabhu","submitted_at":"2024-05-07T16:00:32Z","abstract_excerpt":"PagedAttention is a popular approach for dynamic memory allocation in LLM serving systems. It enables on-demand allocation of GPU memory to mitigate KV cache fragmentation -- a phenomenon that crippled the batch size (and consequently throughput) in prior systems. However, in trying to allocate physical memory at runtime, PagedAttention ends up changing the virtual memory layout of the KV cache from contiguous to non-contiguous. Such a design leads to non-trivial programming and performance overheads.\n  We present vAttention -- an approach that mitigates fragmentation in physical memory while "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.04437","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.04437/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.04437","created_at":"2026-07-05T10:06:40.460525+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.04437v3","created_at":"2026-07-05T10:06:40.460525+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.04437","created_at":"2026-07-05T10:06:40.460525+00:00"},{"alias_kind":"pith_short_12","alias_value":"HFW7PH7UXOGH","created_at":"2026-07-05T10:06:40.460525+00:00"},{"alias_kind":"pith_short_16","alias_value":"HFW7PH7UXOGHQ3UJ","created_at":"2026-07-05T10:06:40.460525+00:00"},{"alias_kind":"pith_short_8","alias_value":"HFW7PH7U","created_at":"2026-07-05T10:06:40.460525+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":95,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26666","citing_title":"PersistentKV: Page-Aware Decode Scheduling for Long-Context LLM Serving on Commodity GPUs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20537","citing_title":"Execution-State Capsules: Graph-Bound Execution-State Checkpoint and Restore for Low-Latency, Small-Batch, On-Device Physical-AI Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26666","citing_title":"PersistentKV: Page-Aware Decode Scheduling for Long-Context LLM Serving on Commodity GPUs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09735","citing_title":"KV-RM: Regularizing KV-Cache Movement for Static-Graph LLM Serving","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15155","citing_title":"eLLM: Elastic Memory Management Framework for Efficient LLM Serving","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09735","citing_title":"KV-RM: Regularizing KV-Cache Movement for Static-Graph LLM Serving","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06036","citing_title":"CodecSight: Leveraging Video Codec Signals for Efficient Streaming VLM Inference","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ","json":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ.json","graph_json":"https://pith.science/api/pith-number/HFW7PH7UXOGHQ3UJDJDGJPABTZ/graph.json","events_json":"https://pith.science/api/pith-number/HFW7PH7UXOGHQ3UJDJDGJPABTZ/events.json","paper":"https://pith.science/paper/HFW7PH7U"},"agent_actions":{"view_html":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ","download_json":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ.json","view_paper":"https://pith.science/paper/HFW7PH7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.04437&json=true","fetch_graph":"https://pith.science/api/pith-number/HFW7PH7UXOGHQ3UJDJDGJPABTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/HFW7PH7UXOGHQ3UJDJDGJPABTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ/action/storage_attestation","attest_author":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ/action/author_attestation","sign_citation":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ/action/citation_signature","submit_replication":"https://pith.science/pith/HFW7PH7UXOGHQ3UJDJDGJPABTZ/action/replication_record"}},"created_at":"2026-07-05T10:06:40.460525+00:00","updated_at":"2026-07-05T10:06:40.460525+00:00"}