{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6FQHDHR6SZN3B6EX3H7PJCRCM7","short_pith_number":"pith:6FQHDHR6","schema_version":"1.0","canonical_sha256":"f160719e3e965bb0f897d9fef48a2267c4f81b73b5bd7ae4c81d198ace7e000d","source":{"kind":"arxiv","id":"2102.03497","version":2},"attestation_state":"computed","paper":{"title":"Weight Rescaling: Effective and Robust Regularization for Deep Neural Networks with Batch Normalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Antoni B. Chan, Jia Wan, Yufei Cui, Yu Mao, Ziquan Liu","submitted_at":"2021-02-06T03:40:20Z","abstract_excerpt":"Weight decay is often used to ensure good generalization in the training practice of deep neural networks with batch normalization (BN-DNNs), where some convolution layers are invariant to weight rescaling due to the normalization. In this paper, we demonstrate that the practical usage of weight decay still has some unsolved problems in spite of existing theoretical work on explaining the effect of weight decay in BN-DNNs. On the one hand, when the non-adaptive learning rate e.g. SGD with momentum is used, the effective learning rate continues to increase even after the initial training stage,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.03497","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-06T03:40:20Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"c5437872a74e34356b1fdf53fa82b38b335fe3e8621a9c3883bff34fe6de60ec","abstract_canon_sha256":"3fb2fbd38e224a34ca6ea460b7e75d4f48ab7430893826c390edbd71db1c0a18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:32:48.780580Z","signature_b64":"s15zCN8+eLX5S3zix0hpYOqyrDUKjs5ETE7Dq1g2uvytZTwvLAR2kjNCr2gphJX9BA3Kth5dWhp0hIxUs+w2BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f160719e3e965bb0f897d9fef48a2267c4f81b73b5bd7ae4c81d198ace7e000d","last_reissued_at":"2026-07-05T04:32:48.780159Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:32:48.780159Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Weight Rescaling: Effective and Robust Regularization for Deep Neural Networks with Batch Normalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Antoni B. Chan, Jia Wan, Yufei Cui, Yu Mao, Ziquan Liu","submitted_at":"2021-02-06T03:40:20Z","abstract_excerpt":"Weight decay is often used to ensure good generalization in the training practice of deep neural networks with batch normalization (BN-DNNs), where some convolution layers are invariant to weight rescaling due to the normalization. In this paper, we demonstrate that the practical usage of weight decay still has some unsolved problems in spite of existing theoretical work on explaining the effect of weight decay in BN-DNNs. On the one hand, when the non-adaptive learning rate e.g. SGD with momentum is used, the effective learning rate continues to increase even after the initial training stage,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.03497","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.03497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.03497","created_at":"2026-07-05T04:32:48.780211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.03497v2","created_at":"2026-07-05T04:32:48.780211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.03497","created_at":"2026-07-05T04:32:48.780211+00:00"},{"alias_kind":"pith_short_12","alias_value":"6FQHDHR6SZN3","created_at":"2026-07-05T04:32:48.780211+00:00"},{"alias_kind":"pith_short_16","alias_value":"6FQHDHR6SZN3B6EX","created_at":"2026-07-05T04:32:48.780211+00:00"},{"alias_kind":"pith_short_8","alias_value":"6FQHDHR6","created_at":"2026-07-05T04:32:48.780211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7","json":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7.json","graph_json":"https://pith.science/api/pith-number/6FQHDHR6SZN3B6EX3H7PJCRCM7/graph.json","events_json":"https://pith.science/api/pith-number/6FQHDHR6SZN3B6EX3H7PJCRCM7/events.json","paper":"https://pith.science/paper/6FQHDHR6"},"agent_actions":{"view_html":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7","download_json":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7.json","view_paper":"https://pith.science/paper/6FQHDHR6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.03497&json=true","fetch_graph":"https://pith.science/api/pith-number/6FQHDHR6SZN3B6EX3H7PJCRCM7/graph.json","fetch_events":"https://pith.science/api/pith-number/6FQHDHR6SZN3B6EX3H7PJCRCM7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7/action/storage_attestation","attest_author":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7/action/author_attestation","sign_citation":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7/action/citation_signature","submit_replication":"https://pith.science/pith/6FQHDHR6SZN3B6EX3H7PJCRCM7/action/replication_record"}},"created_at":"2026-07-05T04:32:48.780211+00:00","updated_at":"2026-07-05T04:32:48.780211+00:00"}