{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:C6XJS2MBEKWJRP3S6LT7P6IND3","short_pith_number":"pith:C6XJS2MB","schema_version":"1.0","canonical_sha256":"17ae99698122ac98bf72f2e7f7f90d1ef9bb17e84b3cc6a20e0fcf86f9336769","source":{"kind":"arxiv","id":"1904.10631","version":2},"attestation_state":"computed","paper":{"title":"Low-Memory Neural Network Training: A Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christopher R. Aberger, Christopher R\\'e, Jian Zhang, Megan Leszczynski, Nimit S. Sohoni","submitted_at":"2019-04-24T03:44:58Z","abstract_excerpt":"Memory is increasingly often the bottleneck when training neural network models. Despite this, techniques to lower the overall memory requirements of training have been less widely studied compared to the extensive literature on reducing the memory requirements of inference. In this paper we study a fundamental question: How much memory is actually needed to train a neural network? To answer this question, we profile the overall memory usage of training on two representative deep learning benchmarks -- the WideResNet model for image classification and the DynamicConv Transformer model for mach"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.10631","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-04-24T03:44:58Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"e02ac9f368e6b4c4ee68df81eae774d8b5351d96d0a6cac19bf3f2aa8d433534","abstract_canon_sha256":"b2afc45c6fd85a75dec440205a78d915e6195812d30269f9e1d568391d21bac2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:12:46.865044Z","signature_b64":"nJyReGXdndMLSHe05KojT2/BL7NNqP5hkDF154yY5W0OJ6DvroWquH0OPquW03yqHqC297T9P4TJem/M8aI0AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"17ae99698122ac98bf72f2e7f7f90d1ef9bb17e84b3cc6a20e0fcf86f9336769","last_reissued_at":"2026-07-05T04:12:46.864617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:12:46.864617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Low-Memory Neural Network Training: A Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christopher R. Aberger, Christopher R\\'e, Jian Zhang, Megan Leszczynski, Nimit S. Sohoni","submitted_at":"2019-04-24T03:44:58Z","abstract_excerpt":"Memory is increasingly often the bottleneck when training neural network models. Despite this, techniques to lower the overall memory requirements of training have been less widely studied compared to the extensive literature on reducing the memory requirements of inference. In this paper we study a fundamental question: How much memory is actually needed to train a neural network? To answer this question, we profile the overall memory usage of training on two representative deep learning benchmarks -- the WideResNet model for image classification and the DynamicConv Transformer model for mach"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.10631","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.10631/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.10631","created_at":"2026-07-05T04:12:46.864679+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.10631v2","created_at":"2026-07-05T04:12:46.864679+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.10631","created_at":"2026-07-05T04:12:46.864679+00:00"},{"alias_kind":"pith_short_12","alias_value":"C6XJS2MBEKWJ","created_at":"2026-07-05T04:12:46.864679+00:00"},{"alias_kind":"pith_short_16","alias_value":"C6XJS2MBEKWJRP3S","created_at":"2026-07-05T04:12:46.864679+00:00"},{"alias_kind":"pith_short_8","alias_value":"C6XJS2MB","created_at":"2026-07-05T04:12:46.864679+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.23640","citing_title":"Geminet: Learning the Duality-based Iterative Process for Lightweight Traffic Engineering in Changing Topologies","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2001.04451","citing_title":"Reformer: The Efficient Transformer","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3","json":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3.json","graph_json":"https://pith.science/api/pith-number/C6XJS2MBEKWJRP3S6LT7P6IND3/graph.json","events_json":"https://pith.science/api/pith-number/C6XJS2MBEKWJRP3S6LT7P6IND3/events.json","paper":"https://pith.science/paper/C6XJS2MB"},"agent_actions":{"view_html":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3","download_json":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3.json","view_paper":"https://pith.science/paper/C6XJS2MB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.10631&json=true","fetch_graph":"https://pith.science/api/pith-number/C6XJS2MBEKWJRP3S6LT7P6IND3/graph.json","fetch_events":"https://pith.science/api/pith-number/C6XJS2MBEKWJRP3S6LT7P6IND3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3/action/storage_attestation","attest_author":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3/action/author_attestation","sign_citation":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3/action/citation_signature","submit_replication":"https://pith.science/pith/C6XJS2MBEKWJRP3S6LT7P6IND3/action/replication_record"}},"created_at":"2026-07-05T04:12:46.864679+00:00","updated_at":"2026-07-05T04:12:46.864679+00:00"}