{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z4I6ECC6ZRHQTP3Y72N4ENZRV4","short_pith_number":"pith:Z4I6ECC6","schema_version":"1.0","canonical_sha256":"cf11e2085ecc4f09bf78fe9bc23731af12153e96d8ca88c5eb99f808ff47b265","source":{"kind":"arxiv","id":"2504.14893","version":1},"attestation_state":"computed","paper":{"title":"Hardware-based Heterogeneous Memory Management for Large Language Model Inference","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Hongbeen Kim, Jaehyuk Huh, Jungwoo Kim, Sanghyeon Lee, Soojin Hwang","submitted_at":"2025-04-21T06:45:41Z","abstract_excerpt":"A large language model (LLM) is one of the most important emerging machine learning applications nowadays. However, due to its huge model size and runtime increase of the memory footprint, LLM inferences suffer from the lack of memory capacity in conventional systems consisting of multiple GPUs with a modest amount of high bandwidth memory. Moreover, since LLM contains many bandwidthintensive kernels, only focusing on the memory capacity without considering the bandwidth incurs a serious performance degradation. To handle such conflicting memory capacity and bandwidth demands in a cost-effecti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14893","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AR","submitted_at":"2025-04-21T06:45:41Z","cross_cats_sorted":[],"title_canon_sha256":"3cec924bd9ee84bed2748bc975d71f9933e75dd6f1c3c8dea987d343e6323bc8","abstract_canon_sha256":"e2768bff00ec14f44d79d544a9906bd634fda1ed254bf361f98ad67c9c2aabe7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:52.398469Z","signature_b64":"aPj3ARDbw68+g/WkFYKy6orJ0jjM2q46tml7jg64JUSmWifPRcr5GYtk0z7Wd0pgRmzJdExvMlGo3QiqFyMtBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf11e2085ecc4f09bf78fe9bc23731af12153e96d8ca88c5eb99f808ff47b265","last_reissued_at":"2026-07-05T10:51:52.397925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:52.397925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hardware-based Heterogeneous Memory Management for Large Language Model Inference","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AR","authors_text":"Hongbeen Kim, Jaehyuk Huh, Jungwoo Kim, Sanghyeon Lee, Soojin Hwang","submitted_at":"2025-04-21T06:45:41Z","abstract_excerpt":"A large language model (LLM) is one of the most important emerging machine learning applications nowadays. However, due to its huge model size and runtime increase of the memory footprint, LLM inferences suffer from the lack of memory capacity in conventional systems consisting of multiple GPUs with a modest amount of high bandwidth memory. Moreover, since LLM contains many bandwidthintensive kernels, only focusing on the memory capacity without considering the bandwidth incurs a serious performance degradation. To handle such conflicting memory capacity and bandwidth demands in a cost-effecti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14893","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14893/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14893","created_at":"2026-07-05T10:51:52.397983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14893v1","created_at":"2026-07-05T10:51:52.397983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14893","created_at":"2026-07-05T10:51:52.397983+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z4I6ECC6ZRHQ","created_at":"2026-07-05T10:51:52.397983+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z4I6ECC6ZRHQTP3Y","created_at":"2026-07-05T10:51:52.397983+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z4I6ECC6","created_at":"2026-07-05T10:51:52.397983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28754","citing_title":"SHIFT: Dynamic Compute Relocation Framework for Communication-Aware Chiplet-Based Systems","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4","json":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4.json","graph_json":"https://pith.science/api/pith-number/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/graph.json","events_json":"https://pith.science/api/pith-number/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/events.json","paper":"https://pith.science/paper/Z4I6ECC6"},"agent_actions":{"view_html":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4","download_json":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4.json","view_paper":"https://pith.science/paper/Z4I6ECC6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14893&json=true","fetch_graph":"https://pith.science/api/pith-number/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/graph.json","fetch_events":"https://pith.science/api/pith-number/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/action/storage_attestation","attest_author":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/action/author_attestation","sign_citation":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/action/citation_signature","submit_replication":"https://pith.science/pith/Z4I6ECC6ZRHQTP3Y72N4ENZRV4/action/replication_record"}},"created_at":"2026-07-05T10:51:52.397983+00:00","updated_at":"2026-07-05T10:51:52.397983+00:00"}