{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5WERKVNC6PBAB2ETNCSQHV7Z47","short_pith_number":"pith:5WERKVNC","schema_version":"1.0","canonical_sha256":"ed891555a2f3c200e89368a503d7f9e7cd03238e645f412f2aab1ff953594dd6","source":{"kind":"arxiv","id":"2408.10013","version":2},"attestation_state":"computed","paper":{"title":"SSDTrain: An Activation Offloading Framework to SSDs for Faster Large Language Model Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.DC","authors_text":"Jeongmin Brian Park, Kun Wu, Mert Hidayeto\\u{g}lu, Sitao Huang, Steven Sam Lumetta, Vikram Sharma Mailthody, Wen-Mei Hwu, Xiaofan Zhang","submitted_at":"2024-08-19T14:09:48Z","abstract_excerpt":"The growth rate of the GPU memory capacity has not been able to keep up with that of the size of large language models (LLMs), hindering the model training process. In particular, activations -- the intermediate tensors produced during forward propagation and reused in backward propagation -- dominate the GPU memory use. This leads to high training overhead such as high weight update cost due to the small micro-batch size. To address this challenge, we propose SSDTrain, an adaptive activation offloading framework to high-capacity NVMe SSDs. SSDTrain reduces GPU memory usage without impacting p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.10013","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-08-19T14:09:48Z","cross_cats_sorted":["cs.NE"],"title_canon_sha256":"4b7b8886bd77c5cf08024926d794ab0a408dc53e07defe51fc6879bff4ff334a","abstract_canon_sha256":"f4560983f32d329ff9e980df225474fd85fea0b9a8a38a5a9a35758f1fb125e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:44.530108Z","signature_b64":"6fGOwkV2R3uJGCKhFNe5UHB0XN08bIbx2BAPK15nQpH22WpYBd7pNxhRUS9YM2rSqTQCROPz3r6nJvkOHh/qBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed891555a2f3c200e89368a503d7f9e7cd03238e645f412f2aab1ff953594dd6","last_reissued_at":"2026-07-05T10:14:44.529561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:44.529561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SSDTrain: An Activation Offloading Framework to SSDs for Faster Large Language Model Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.DC","authors_text":"Jeongmin Brian Park, Kun Wu, Mert Hidayeto\\u{g}lu, Sitao Huang, Steven Sam Lumetta, Vikram Sharma Mailthody, Wen-Mei Hwu, Xiaofan Zhang","submitted_at":"2024-08-19T14:09:48Z","abstract_excerpt":"The growth rate of the GPU memory capacity has not been able to keep up with that of the size of large language models (LLMs), hindering the model training process. In particular, activations -- the intermediate tensors produced during forward propagation and reused in backward propagation -- dominate the GPU memory use. This leads to high training overhead such as high weight update cost due to the small micro-batch size. To address this challenge, we propose SSDTrain, an adaptive activation offloading framework to high-capacity NVMe SSDs. SSDTrain reduces GPU memory usage without impacting p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.10013","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.10013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.10013","created_at":"2026-07-05T10:14:44.529619+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.10013v2","created_at":"2026-07-05T10:14:44.529619+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.10013","created_at":"2026-07-05T10:14:44.529619+00:00"},{"alias_kind":"pith_short_12","alias_value":"5WERKVNC6PBA","created_at":"2026-07-05T10:14:44.529619+00:00"},{"alias_kind":"pith_short_16","alias_value":"5WERKVNC6PBAB2ET","created_at":"2026-07-05T10:14:44.529619+00:00"},{"alias_kind":"pith_short_8","alias_value":"5WERKVNC","created_at":"2026-07-05T10:14:44.529619+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.01143","citing_title":"HexiScale: Facilitating Large Language Model Training over Heterogeneous Hardware","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47","json":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47.json","graph_json":"https://pith.science/api/pith-number/5WERKVNC6PBAB2ETNCSQHV7Z47/graph.json","events_json":"https://pith.science/api/pith-number/5WERKVNC6PBAB2ETNCSQHV7Z47/events.json","paper":"https://pith.science/paper/5WERKVNC"},"agent_actions":{"view_html":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47","download_json":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47.json","view_paper":"https://pith.science/paper/5WERKVNC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.10013&json=true","fetch_graph":"https://pith.science/api/pith-number/5WERKVNC6PBAB2ETNCSQHV7Z47/graph.json","fetch_events":"https://pith.science/api/pith-number/5WERKVNC6PBAB2ETNCSQHV7Z47/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47/action/storage_attestation","attest_author":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47/action/author_attestation","sign_citation":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47/action/citation_signature","submit_replication":"https://pith.science/pith/5WERKVNC6PBAB2ETNCSQHV7Z47/action/replication_record"}},"created_at":"2026-07-05T10:14:44.529619+00:00","updated_at":"2026-07-05T10:14:44.529619+00:00"}