{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VIQFJFILCTKBI7YHPJIICWZ2WV","short_pith_number":"pith:VIQFJFIL","schema_version":"1.0","canonical_sha256":"aa2054950b14d4147f077a50815b3ab56c65e8d855129e015e2da1c5b4b29dd3","source":{"kind":"arxiv","id":"2411.01142","version":1},"attestation_state":"computed","paper":{"title":"NEO: Saving GPU Memory Crisis with CPU Offloading for Online LLM Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Ion Stoica, Minlan Yu, Shiyi Cao, Xuanlin Jiang, Yang Zhou","submitted_at":"2024-11-02T05:15:44Z","abstract_excerpt":"Online LLM inference powers many exciting applications such as intelligent chatbots and autonomous agents. Modern LLM inference engines widely rely on request batching to improve inference throughput, aiming to make it cost-efficient when running on expensive GPU accelerators. However, the limited GPU memory has largely limited the batch size achieved in practice, leaving significant GPU compute resources wasted.\n  We present NEO, an online LLM inference system that offloads part of attention compute and KV cache states from the GPU to the local host CPU, effectively increasing the GPU batch s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01142","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-11-02T05:15:44Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8ead6b048cd493d90ba1aaf7dcfaf01251bf9f001bfd473bc8fa6e68fa3348f3","abstract_canon_sha256":"ffc3a3e29947ab9a34e6e5367c13f606f99ecfa630641fc7ba312399b73475bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:25.308274Z","signature_b64":"KqRtkgrE+dWuedLzQTAihuKKTtyJuSGpKnnNqmxGOdnKQB2A5qDpqBh10SFqxJIU2WLbE7Tlzf5aI4iTjhgqCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa2054950b14d4147f077a50815b3ab56c65e8d855129e015e2da1c5b4b29dd3","last_reissued_at":"2026-07-05T09:30:25.307841Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:25.307841Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NEO: Saving GPU Memory Crisis with CPU Offloading for Online LLM Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Ion Stoica, Minlan Yu, Shiyi Cao, Xuanlin Jiang, Yang Zhou","submitted_at":"2024-11-02T05:15:44Z","abstract_excerpt":"Online LLM inference powers many exciting applications such as intelligent chatbots and autonomous agents. Modern LLM inference engines widely rely on request batching to improve inference throughput, aiming to make it cost-efficient when running on expensive GPU accelerators. However, the limited GPU memory has largely limited the batch size achieved in practice, leaving significant GPU compute resources wasted.\n  We present NEO, an online LLM inference system that offloads part of attention compute and KV cache states from the GPU to the local host CPU, effectively increasing the GPU batch s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01142","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01142/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01142","created_at":"2026-07-05T09:30:25.307891+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01142v1","created_at":"2026-07-05T09:30:25.307891+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01142","created_at":"2026-07-05T09:30:25.307891+00:00"},{"alias_kind":"pith_short_12","alias_value":"VIQFJFILCTKB","created_at":"2026-07-05T09:30:25.307891+00:00"},{"alias_kind":"pith_short_16","alias_value":"VIQFJFILCTKBI7YH","created_at":"2026-07-05T09:30:25.307891+00:00"},{"alias_kind":"pith_short_8","alias_value":"VIQFJFIL","created_at":"2026-07-05T09:30:25.307891+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.00868","citing_title":"FlexiCache: Leveraging Temporal Stability of Attention Heads for Efficient KV Cache Management","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21026","citing_title":"MCAP: Deployment-Time Layer Profiling for Memory-Constrained LLM Inference","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV","json":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV.json","graph_json":"https://pith.science/api/pith-number/VIQFJFILCTKBI7YHPJIICWZ2WV/graph.json","events_json":"https://pith.science/api/pith-number/VIQFJFILCTKBI7YHPJIICWZ2WV/events.json","paper":"https://pith.science/paper/VIQFJFIL"},"agent_actions":{"view_html":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV","download_json":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV.json","view_paper":"https://pith.science/paper/VIQFJFIL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01142&json=true","fetch_graph":"https://pith.science/api/pith-number/VIQFJFILCTKBI7YHPJIICWZ2WV/graph.json","fetch_events":"https://pith.science/api/pith-number/VIQFJFILCTKBI7YHPJIICWZ2WV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV/action/storage_attestation","attest_author":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV/action/author_attestation","sign_citation":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV/action/citation_signature","submit_replication":"https://pith.science/pith/VIQFJFILCTKBI7YHPJIICWZ2WV/action/replication_record"}},"created_at":"2026-07-05T09:30:25.307891+00:00","updated_at":"2026-07-05T09:30:25.307891+00:00"}