{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:V6CNNA62EQUP2E4GIPCTLCVYTF","short_pith_number":"pith:V6CNNA62","schema_version":"1.0","canonical_sha256":"af84d683da2428fd138643c5358ab8997dbc5033c699a428ac49591572a33fe4","source":{"kind":"arxiv","id":"1811.03600","version":3},"attestation_state":"computed","paper":{"title":"Measuring the Effects of Data Parallelism on Neural Network Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christopher J. Shallue, George E. Dahl, Jaehoon Lee, Jascha Sohl-Dickstein, Joseph Antognini, Roy Frostig","submitted_at":"2018-11-08T18:33:41Z","abstract_excerpt":"Recent hardware developments have dramatically increased the scale of data parallelism available for neural network training. Among the simplest ways to harness next-generation hardware is to increase the batch size in standard mini-batch neural network training algorithms. In this work, we aim to experimentally characterize the effects of increasing the batch size on training time, as measured by the number of steps necessary to reach a goal out-of-sample error. We study how this relationship varies with the training algorithm, model, and data set, and find extremely large variation between w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1811.03600","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-11-08T18:33:41Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"87c9b0b90d6e6dccdc89efa87f504f0c1fb4c23813f1649f41dc06388b38c93e","abstract_canon_sha256":"0ee94773d8b6409a453d0a2dee0807d4c324aef47147aa89d0714a2f9ed70de8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:40:11.471482Z","signature_b64":"1dyVD30LRLPjp5Ld3hHmQk8Vwvm5c85bQBLtmHWr6yRM+Y+W1KXNQGV/JnjkRHxZ1r709JTzqWLZR8h+yW+7DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af84d683da2428fd138643c5358ab8997dbc5033c699a428ac49591572a33fe4","last_reissued_at":"2026-05-17T23:40:11.470536Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:40:11.470536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring the Effects of Data Parallelism on Neural Network Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christopher J. Shallue, George E. Dahl, Jaehoon Lee, Jascha Sohl-Dickstein, Joseph Antognini, Roy Frostig","submitted_at":"2018-11-08T18:33:41Z","abstract_excerpt":"Recent hardware developments have dramatically increased the scale of data parallelism available for neural network training. Among the simplest ways to harness next-generation hardware is to increase the batch size in standard mini-batch neural network training algorithms. In this work, we aim to experimentally characterize the effects of increasing the batch size on training time, as measured by the number of steps necessary to reach a goal out-of-sample error. We study how this relationship varies with the training algorithm, model, and data set, and find extremely large variation between w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1811.03600","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1811.03600","created_at":"2026-05-17T23:40:11.470708+00:00"},{"alias_kind":"arxiv_version","alias_value":"1811.03600v3","created_at":"2026-05-17T23:40:11.470708+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1811.03600","created_at":"2026-05-17T23:40:11.470708+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6CNNA62EQUP","created_at":"2026-05-18T12:32:59.047623+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6CNNA62EQUP2E4G","created_at":"2026-05-18T12:32:59.047623+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6CNNA62","created_at":"2026-05-18T12:32:59.047623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":6,"sample":[{"citing_arxiv_id":"1906.11786","citing_title":"Fast Training of Sparse Graph Neural Networks on Dense Hardware","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"1904.00962","citing_title":"Large Batch Optimization for Deep Learning: Training BERT in 76 minutes","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2102.01293","citing_title":"Scaling Laws for Transfer","ref_index":106,"is_internal_anchor":true},{"citing_arxiv_id":"1909.06335","citing_title":"Measuring the Effects of Non-Identical Data Distribution for Federated Visual Classification","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"1908.10063","citing_title":"FinBERT: Financial Sentiment Analysis with Pre-trained Language Models","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2605.14200","citing_title":"How to Scale Mixture-of-Experts: From muP to the Maximally Scale-Stable Parameterization","ref_index":70,"is_internal_anchor":true},{"citing_arxiv_id":"1910.10683","citing_title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21215","citing_title":"The Recurrent Transformer: Greater Effective Depth and Efficient Decoding","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05221","citing_title":"Language Models (Mostly) Know What They Know","ref_index":225,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF","json":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF.json","graph_json":"https://pith.science/api/pith-number/V6CNNA62EQUP2E4GIPCTLCVYTF/graph.json","events_json":"https://pith.science/api/pith-number/V6CNNA62EQUP2E4GIPCTLCVYTF/events.json","paper":"https://pith.science/paper/V6CNNA62"},"agent_actions":{"view_html":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF","download_json":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF.json","view_paper":"https://pith.science/paper/V6CNNA62","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1811.03600&json=true","fetch_graph":"https://pith.science/api/pith-number/V6CNNA62EQUP2E4GIPCTLCVYTF/graph.json","fetch_events":"https://pith.science/api/pith-number/V6CNNA62EQUP2E4GIPCTLCVYTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF/action/storage_attestation","attest_author":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF/action/author_attestation","sign_citation":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF/action/citation_signature","submit_replication":"https://pith.science/pith/V6CNNA62EQUP2E4GIPCTLCVYTF/action/replication_record"}},"created_at":"2026-05-17T23:40:11.470708+00:00","updated_at":"2026-05-17T23:40:11.470708+00:00"}