{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:2RTYD45DJLPJ4XFGJYD6RSBDFP","short_pith_number":"pith:2RTYD45D","schema_version":"1.0","canonical_sha256":"d46781f3a34ade9e5ca64e07e8c8232bc80830aa3a58867f7ca19bfe532defbe","source":{"kind":"arxiv","id":"1903.10520","version":2},"attestation_state":"computed","paper":{"title":"Micro-Batch Training with Batch-Channel Normalization and Weight Standardization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Chenxi Liu, Huiyu Wang, Siyuan Qiao, Wei Shen","submitted_at":"2019-03-25T18:00:05Z","abstract_excerpt":"Batch Normalization (BN) has become an out-of-box technique to improve deep network training. However, its effectiveness is limited for micro-batch training, i.e., each GPU typically has only 1-2 images for training, which is inevitable for many computer vision tasks, e.g., object detection and semantic segmentation, constrained by memory consumption. To address this issue, we propose Weight Standardization (WS) and Batch-Channel Normalization (BCN) to bring two success factors of BN into micro-batch training: 1) the smoothing effects on the loss landscape and 2) the ability to avoid harmful e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1903.10520","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-03-25T18:00:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"222d1750d6b620eba227750147a281f4ad45cc815aa5c977cf3c8860d8afc717","abstract_canon_sha256":"be425a20fef11e761120a85a1b93029e4222fc2b4df227d089f797fdc518072f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:25:34.142296Z","signature_b64":"1Vkdv4R3VzuoYwrxLc6WiiVIfiM8nU9jExDANbzF6uuFs3Xj06BPY0EwRj9pE6ErQHbFEqB7JNWqtc1G1QwxAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d46781f3a34ade9e5ca64e07e8c8232bc80830aa3a58867f7ca19bfe532defbe","last_reissued_at":"2026-07-05T01:25:34.141878Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:25:34.141878Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Micro-Batch Training with Batch-Channel Normalization and Weight Standardization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Chenxi Liu, Huiyu Wang, Siyuan Qiao, Wei Shen","submitted_at":"2019-03-25T18:00:05Z","abstract_excerpt":"Batch Normalization (BN) has become an out-of-box technique to improve deep network training. However, its effectiveness is limited for micro-batch training, i.e., each GPU typically has only 1-2 images for training, which is inevitable for many computer vision tasks, e.g., object detection and semantic segmentation, constrained by memory consumption. To address this issue, we propose Weight Standardization (WS) and Batch-Channel Normalization (BCN) to bring two success factors of BN into micro-batch training: 1) the smoothing effects on the loss landscape and 2) the ability to avoid harmful e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1903.10520","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1903.10520/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1903.10520","created_at":"2026-07-05T01:25:34.141940+00:00"},{"alias_kind":"arxiv_version","alias_value":"1903.10520v2","created_at":"2026-07-05T01:25:34.141940+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1903.10520","created_at":"2026-07-05T01:25:34.141940+00:00"},{"alias_kind":"pith_short_12","alias_value":"2RTYD45DJLPJ","created_at":"2026-07-05T01:25:34.141940+00:00"},{"alias_kind":"pith_short_16","alias_value":"2RTYD45DJLPJ4XFG","created_at":"2026-07-05T01:25:34.141940+00:00"},{"alias_kind":"pith_short_8","alias_value":"2RTYD45D","created_at":"2026-07-05T01:25:34.141940+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06470","citing_title":"PC Layer: Polynomial Weight Preconditioning for Improving LLM Pre-Training","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31695","citing_title":"Intrinsically Stable Spiking Neural Networks: Overcoming the Performance Barrier in the Absence of Batch Normalization","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30926","citing_title":"SpikON: A Dual-Parallel and Efficient Accelerator for Online Spiking Neural Networks Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04418","citing_title":"Demystifying Manifold Constraints in LLM Pre-training","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP","json":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP.json","graph_json":"https://pith.science/api/pith-number/2RTYD45DJLPJ4XFGJYD6RSBDFP/graph.json","events_json":"https://pith.science/api/pith-number/2RTYD45DJLPJ4XFGJYD6RSBDFP/events.json","paper":"https://pith.science/paper/2RTYD45D"},"agent_actions":{"view_html":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP","download_json":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP.json","view_paper":"https://pith.science/paper/2RTYD45D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1903.10520&json=true","fetch_graph":"https://pith.science/api/pith-number/2RTYD45DJLPJ4XFGJYD6RSBDFP/graph.json","fetch_events":"https://pith.science/api/pith-number/2RTYD45DJLPJ4XFGJYD6RSBDFP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP/action/storage_attestation","attest_author":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP/action/author_attestation","sign_citation":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP/action/citation_signature","submit_replication":"https://pith.science/pith/2RTYD45DJLPJ4XFGJYD6RSBDFP/action/replication_record"}},"created_at":"2026-07-05T01:25:34.141940+00:00","updated_at":"2026-07-05T01:25:34.141940+00:00"}