{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AYJ4LRYXCN52YVH34PWFVLINUQ","short_pith_number":"pith:AYJ4LRYX","schema_version":"1.0","canonical_sha256":"0613c5c717137bac54fbe3ec5aad0da41c154013e688403ce1b056cc8aab86a3","source":{"kind":"arxiv","id":"2502.16963","version":1},"attestation_state":"computed","paper":{"title":"Make LLM Inference Affordable to Everyone: Augmenting GPU Memory with NDP-DIMM","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Bing Li, Haimeng Ren, Lian Liu, Mengdi Wang, Shixin Zhao, Xiaowei Li, Ying Wang, Yinhe Han, Zhaohui Xu","submitted_at":"2025-02-24T08:41:19Z","abstract_excerpt":"The billion-scale Large Language Models (LLMs) need deployment on expensive server-grade GPUs with large-storage HBMs and abundant computation capability. As LLM-assisted services become popular, achieving cost-effective LLM inference on budget-friendly hardware becomes the trend. Extensive researches relocate LLM parameters from expensive GPUs to host memory. However, the restricted bandwidth between the host and GPU memory limits the inference performance.\n  This work introduces Hermes, a budget-friendly system that leverages the near-data processing (NDP) within commodity DRAM DIMMs to enha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16963","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AR","submitted_at":"2025-02-24T08:41:19Z","cross_cats_sorted":[],"title_canon_sha256":"4583b8c0c3c6f673e8cc788253ead3d8589b7283bcd431fa3133ebc31a2e6fb2","abstract_canon_sha256":"88405b49d111d704f5dbca2defdd884fcabbb22773fdf8d449087dc2318f6883"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:04.016992Z","signature_b64":"CTAGgZCXDC2sDeZO09joqOkamKdBBLAPRS5YF+MX70b84Us0f6cqbOGmNKVfYXqIOLnsIkRhs8Iz8jb44e4sBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0613c5c717137bac54fbe3ec5aad0da41c154013e688403ce1b056cc8aab86a3","last_reissued_at":"2026-07-05T10:19:04.016470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:04.016470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Make LLM Inference Affordable to Everyone: Augmenting GPU Memory with NDP-DIMM","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Bing Li, Haimeng Ren, Lian Liu, Mengdi Wang, Shixin Zhao, Xiaowei Li, Ying Wang, Yinhe Han, Zhaohui Xu","submitted_at":"2025-02-24T08:41:19Z","abstract_excerpt":"The billion-scale Large Language Models (LLMs) need deployment on expensive server-grade GPUs with large-storage HBMs and abundant computation capability. As LLM-assisted services become popular, achieving cost-effective LLM inference on budget-friendly hardware becomes the trend. Extensive researches relocate LLM parameters from expensive GPUs to host memory. However, the restricted bandwidth between the host and GPU memory limits the inference performance.\n  This work introduces Hermes, a budget-friendly system that leverages the near-data processing (NDP) within commodity DRAM DIMMs to enha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16963","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16963","created_at":"2026-07-05T10:19:04.016533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16963v1","created_at":"2026-07-05T10:19:04.016533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16963","created_at":"2026-07-05T10:19:04.016533+00:00"},{"alias_kind":"pith_short_12","alias_value":"AYJ4LRYXCN52","created_at":"2026-07-05T10:19:04.016533+00:00"},{"alias_kind":"pith_short_16","alias_value":"AYJ4LRYXCN52YVH3","created_at":"2026-07-05T10:19:04.016533+00:00"},{"alias_kind":"pith_short_8","alias_value":"AYJ4LRYX","created_at":"2026-07-05T10:19:04.016533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.17584","citing_title":"L3: DIMM-PIM Integrated Architecture and Coordination for Scalable Long-Context LLM Inference","ref_index":77,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ","json":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ.json","graph_json":"https://pith.science/api/pith-number/AYJ4LRYXCN52YVH34PWFVLINUQ/graph.json","events_json":"https://pith.science/api/pith-number/AYJ4LRYXCN52YVH34PWFVLINUQ/events.json","paper":"https://pith.science/paper/AYJ4LRYX"},"agent_actions":{"view_html":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ","download_json":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ.json","view_paper":"https://pith.science/paper/AYJ4LRYX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16963&json=true","fetch_graph":"https://pith.science/api/pith-number/AYJ4LRYXCN52YVH34PWFVLINUQ/graph.json","fetch_events":"https://pith.science/api/pith-number/AYJ4LRYXCN52YVH34PWFVLINUQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ/action/storage_attestation","attest_author":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ/action/author_attestation","sign_citation":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ/action/citation_signature","submit_replication":"https://pith.science/pith/AYJ4LRYXCN52YVH34PWFVLINUQ/action/replication_record"}},"created_at":"2026-07-05T10:19:04.016533+00:00","updated_at":"2026-07-05T10:19:04.016533+00:00"}