{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DWL4V6KABJUCNMLFEIXYWM4GMM","short_pith_number":"pith:DWL4V6KA","schema_version":"1.0","canonical_sha256":"1d97caf9400a6826b165222f8b33866334bff32a3af12d68f20869fab33aa3bd","source":{"kind":"arxiv","id":"2502.08182","version":1},"attestation_state":"computed","paper":{"title":"Memory Offloading for Large Language Model Inference with Latency SLO Guarantees","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chenxiang Ma, Diyu Zhou, Hanyu Zhao, Jiaxun Han, Jie Zhang, Tianhao Fu, Xiaolin Wang, Yingwei Luo, Yong Li, Zehua Yang, Zhenlin Wang, Zhisheng Ye","submitted_at":"2025-02-12T07:42:45Z","abstract_excerpt":"Offloading large language models (LLMs) state to host memory during inference promises to reduce operational costs by supporting larger models, longer inputs, and larger batch sizes. However, the design of existing memory offloading mechanisms does not take latency service-level objectives (SLOs) into consideration. As a result, they either lead to frequent SLO violations or underutilize host memory, thereby incurring economic loss and thus defeating the purpose of memory offloading.\n  This paper presents Select-N, a latency-SLO-aware memory offloading system for LLM serving. A key challenge i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08182","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-02-12T07:42:45Z","cross_cats_sorted":[],"title_canon_sha256":"3e8df8376810fc22ae0783f1f3ffc764c452abbbcddcab426ca21af19ed029a2","abstract_canon_sha256":"181265903e190f4e6d5109d16b68891fa125ff54b745f88586505f5bd0a5ebff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:18.381784Z","signature_b64":"f+NBzgKD2TPlYlJYPgT0rG4FWE0eH3/7eMEt6lkmpNTBjaWMkKPCFuc0ewTTuOaqfFGBgzbCDJd28UVONMYaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d97caf9400a6826b165222f8b33866334bff32a3af12d68f20869fab33aa3bd","last_reissued_at":"2026-07-05T10:13:18.381209Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:18.381209Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memory Offloading for Large Language Model Inference with Latency SLO Guarantees","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chenxiang Ma, Diyu Zhou, Hanyu Zhao, Jiaxun Han, Jie Zhang, Tianhao Fu, Xiaolin Wang, Yingwei Luo, Yong Li, Zehua Yang, Zhenlin Wang, Zhisheng Ye","submitted_at":"2025-02-12T07:42:45Z","abstract_excerpt":"Offloading large language models (LLMs) state to host memory during inference promises to reduce operational costs by supporting larger models, longer inputs, and larger batch sizes. However, the design of existing memory offloading mechanisms does not take latency service-level objectives (SLOs) into consideration. As a result, they either lead to frequent SLO violations or underutilize host memory, thereby incurring economic loss and thus defeating the purpose of memory offloading.\n  This paper presents Select-N, a latency-SLO-aware memory offloading system for LLM serving. A key challenge i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08182","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08182/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08182","created_at":"2026-07-05T10:13:18.381270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08182v1","created_at":"2026-07-05T10:13:18.381270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08182","created_at":"2026-07-05T10:13:18.381270+00:00"},{"alias_kind":"pith_short_12","alias_value":"DWL4V6KABJUC","created_at":"2026-07-05T10:13:18.381270+00:00"},{"alias_kind":"pith_short_16","alias_value":"DWL4V6KABJUCNMLF","created_at":"2026-07-05T10:13:18.381270+00:00"},{"alias_kind":"pith_short_8","alias_value":"DWL4V6KA","created_at":"2026-07-05T10:13:18.381270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.20309","citing_title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM","json":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM.json","graph_json":"https://pith.science/api/pith-number/DWL4V6KABJUCNMLFEIXYWM4GMM/graph.json","events_json":"https://pith.science/api/pith-number/DWL4V6KABJUCNMLFEIXYWM4GMM/events.json","paper":"https://pith.science/paper/DWL4V6KA"},"agent_actions":{"view_html":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM","download_json":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM.json","view_paper":"https://pith.science/paper/DWL4V6KA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08182&json=true","fetch_graph":"https://pith.science/api/pith-number/DWL4V6KABJUCNMLFEIXYWM4GMM/graph.json","fetch_events":"https://pith.science/api/pith-number/DWL4V6KABJUCNMLFEIXYWM4GMM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM/action/storage_attestation","attest_author":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM/action/author_attestation","sign_citation":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM/action/citation_signature","submit_replication":"https://pith.science/pith/DWL4V6KABJUCNMLFEIXYWM4GMM/action/replication_record"}},"created_at":"2026-07-05T10:13:18.381270+00:00","updated_at":"2026-07-05T10:13:18.381270+00:00"}