{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:FRNB3XM7XQTO5XLFCQ2UNSFLT7","short_pith_number":"pith:FRNB3XM7","schema_version":"1.0","canonical_sha256":"2c5a1ddd9fbc26eedd65143546c8ab9ff08be33b1df3a79191b1e464b35edd7f","source":{"kind":"arxiv","id":"2103.15345","version":1},"attestation_state":"computed","paper":{"title":"FixNorm: Dissecting Weight Decay for Training Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Yucong Zhou, Yunxiao Sun, Zhao Zhong","submitted_at":"2021-03-29T05:41:56Z","abstract_excerpt":"Weight decay is a widely used technique for training Deep Neural Networks(DNN). It greatly affects generalization performance but the underlying mechanisms are not fully understood. Recent works show that for layers followed by normalizations, weight decay mainly affects the effective learning rate. However, despite normalizations have been extensively adopted in modern DNNs, layers such as the final fully-connected layer do not satisfy this precondition. For these layers, the effects of weight decay are still unclear. In this paper, we comprehensively investigate the mechanisms of weight deca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.15345","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-29T05:41:56Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"8c1f6b1e01d21469c00f88c178995594d7d9b9d0bb5407c0a6fdf7003f4cab9f","abstract_canon_sha256":"d1291a5ceb59c90a73f3ee821abf21a624d08a245a0887ef1f660dc91f253055"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:27:05.129604Z","signature_b64":"9WB8yoIFkR25xI0JnF7+yXlJlfLnzZIHXWE0rhMYKV2nuZf3xvFKwnk4frszUBwirsU92ArHPrRYHgROa/XFDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c5a1ddd9fbc26eedd65143546c8ab9ff08be33b1df3a79191b1e464b35edd7f","last_reissued_at":"2026-07-05T02:27:05.129091Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:27:05.129091Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FixNorm: Dissecting Weight Decay for Training Deep Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Yucong Zhou, Yunxiao Sun, Zhao Zhong","submitted_at":"2021-03-29T05:41:56Z","abstract_excerpt":"Weight decay is a widely used technique for training Deep Neural Networks(DNN). It greatly affects generalization performance but the underlying mechanisms are not fully understood. Recent works show that for layers followed by normalizations, weight decay mainly affects the effective learning rate. However, despite normalizations have been extensively adopted in modern DNNs, layers such as the final fully-connected layer do not satisfy this precondition. For these layers, the effects of weight decay are still unclear. In this paper, we comprehensively investigate the mechanisms of weight deca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.15345","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.15345/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.15345","created_at":"2026-07-05T02:27:05.129152+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.15345v1","created_at":"2026-07-05T02:27:05.129152+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.15345","created_at":"2026-07-05T02:27:05.129152+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRNB3XM7XQTO","created_at":"2026-07-05T02:27:05.129152+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRNB3XM7XQTO5XLF","created_at":"2026-07-05T02:27:05.129152+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRNB3XM7","created_at":"2026-07-05T02:27:05.129152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7","json":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7.json","graph_json":"https://pith.science/api/pith-number/FRNB3XM7XQTO5XLFCQ2UNSFLT7/graph.json","events_json":"https://pith.science/api/pith-number/FRNB3XM7XQTO5XLFCQ2UNSFLT7/events.json","paper":"https://pith.science/paper/FRNB3XM7"},"agent_actions":{"view_html":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7","download_json":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7.json","view_paper":"https://pith.science/paper/FRNB3XM7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.15345&json=true","fetch_graph":"https://pith.science/api/pith-number/FRNB3XM7XQTO5XLFCQ2UNSFLT7/graph.json","fetch_events":"https://pith.science/api/pith-number/FRNB3XM7XQTO5XLFCQ2UNSFLT7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7/action/storage_attestation","attest_author":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7/action/author_attestation","sign_citation":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7/action/citation_signature","submit_replication":"https://pith.science/pith/FRNB3XM7XQTO5XLFCQ2UNSFLT7/action/replication_record"}},"created_at":"2026-07-05T02:27:05.129152+00:00","updated_at":"2026-07-05T02:27:05.129152+00:00"}