{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:25X6ZO74XWYHAYEJ3JIVQ64YAW","short_pith_number":"pith:25X6ZO74","schema_version":"1.0","canonical_sha256":"d76fecbbfcbdb0706089da51587b9805bc5637dc626aa9aa23d57813cbbd99c0","source":{"kind":"arxiv","id":"2412.21124","version":2},"attestation_state":"computed","paper":{"title":"Adaptive Batch Size Schedules for Distributed Training of Language Models with Data and Model Parallelism","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chenwei Xu, Han Liu, Mladen Kolar, Tim Tsz-Kit Lau, Weijian Li","submitted_at":"2024-12-30T17:55:28Z","abstract_excerpt":"An appropriate choice of batch sizes in large-scale model training is crucial, yet it involves an intrinsic yet inevitable dilemma: large-batch training improves training efficiency in terms of memory utilization, while generalization performance often deteriorates due to small amounts of gradient noise. Despite this dilemma, the common practice of choosing batch sizes in language model training often prioritizes training efficiency -- employing either constant large sizes with data parallelism or implementing batch size warmup schedules. However, such batch size schedule designs remain heuris"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.21124","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-30T17:55:28Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"4621507e37f9afdde61b9056ef6d21e3ee21dc1b1114f231be4e2dd78449267d","abstract_canon_sha256":"bdf52e51bb146d2928ec0090580003154b70ee952f7288db6b946d96add21b30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:52.360443Z","signature_b64":"e7Q4coUBZheoMNEgjvDQMBKTrw0Tg7sMN0eTRaWBjhvJ1HwqiW4ZszZed65OAtkbtGH+0wlr+vCxnkV7SGujDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d76fecbbfcbdb0706089da51587b9805bc5637dc626aa9aa23d57813cbbd99c0","last_reissued_at":"2026-07-05T10:31:52.359930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:52.359930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Batch Size Schedules for Distributed Training of Language Models with Data and Model Parallelism","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chenwei Xu, Han Liu, Mladen Kolar, Tim Tsz-Kit Lau, Weijian Li","submitted_at":"2024-12-30T17:55:28Z","abstract_excerpt":"An appropriate choice of batch sizes in large-scale model training is crucial, yet it involves an intrinsic yet inevitable dilemma: large-batch training improves training efficiency in terms of memory utilization, while generalization performance often deteriorates due to small amounts of gradient noise. Despite this dilemma, the common practice of choosing batch sizes in language model training often prioritizes training efficiency -- employing either constant large sizes with data parallelism or implementing batch size warmup schedules. However, such batch size schedule designs remain heuris"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.21124","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.21124/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.21124","created_at":"2026-07-05T10:31:52.359982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.21124v2","created_at":"2026-07-05T10:31:52.359982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.21124","created_at":"2026-07-05T10:31:52.359982+00:00"},{"alias_kind":"pith_short_12","alias_value":"25X6ZO74XWYH","created_at":"2026-07-05T10:31:52.359982+00:00"},{"alias_kind":"pith_short_16","alias_value":"25X6ZO74XWYHAYEJ","created_at":"2026-07-05T10:31:52.359982+00:00"},{"alias_kind":"pith_short_8","alias_value":"25X6ZO74","created_at":"2026-07-05T10:31:52.359982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.06350","citing_title":"Convergence of Riemannian Stochastic Gradient Descents: Varying Batch Sizes And Nonstandard Batch Forming","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW","json":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW.json","graph_json":"https://pith.science/api/pith-number/25X6ZO74XWYHAYEJ3JIVQ64YAW/graph.json","events_json":"https://pith.science/api/pith-number/25X6ZO74XWYHAYEJ3JIVQ64YAW/events.json","paper":"https://pith.science/paper/25X6ZO74"},"agent_actions":{"view_html":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW","download_json":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW.json","view_paper":"https://pith.science/paper/25X6ZO74","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.21124&json=true","fetch_graph":"https://pith.science/api/pith-number/25X6ZO74XWYHAYEJ3JIVQ64YAW/graph.json","fetch_events":"https://pith.science/api/pith-number/25X6ZO74XWYHAYEJ3JIVQ64YAW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW/action/storage_attestation","attest_author":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW/action/author_attestation","sign_citation":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW/action/citation_signature","submit_replication":"https://pith.science/pith/25X6ZO74XWYHAYEJ3JIVQ64YAW/action/replication_record"}},"created_at":"2026-07-05T10:31:52.359982+00:00","updated_at":"2026-07-05T10:31:52.359982+00:00"}