{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:2VVFABLW6MRS7E2FDUFPFKU4OL","short_pith_number":"pith:2VVFABLW","schema_version":"1.0","canonical_sha256":"d56a500576f3232f93451d0af2aa9c72cd19834268ec8b2cf78c3a18473965f7","source":{"kind":"arxiv","id":"2102.06171","version":1},"attestation_state":"computed","paper":{"title":"High-Performance Large-Scale Image Recognition Without Normalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Andrew Brock, Karen Simonyan, Samuel L. Smith, Soham De","submitted_at":"2021-02-11T18:23:20Z","abstract_excerpt":"Batch normalization is a key component of most image classification models, but it has many undesirable properties stemming from its dependence on the batch size and interactions between examples. Although recent work has succeeded in training deep ResNets without normalization layers, these models do not match the test accuracies of the best batch-normalized networks, and are often unstable for large learning rates or strong data augmentations. In this work, we develop an adaptive gradient clipping technique which overcomes these instabilities, and design a significantly improved class of Nor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.06171","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-02-11T18:23:20Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"fc452a76dc3de427f426e682fc6f2f57f655299e82e5ad6a79d466b508b979f7","abstract_canon_sha256":"13c4f3907d9dd591b51726b8a3c16802baf5b2b396e4c0ac38ad9c1ed21f8b28"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:14:39.063524Z","signature_b64":"3+HNDETVs6yfAKdjlRkNyQV7guaZWROlqn4wDn2uSjVS/qfMtzQ1YFfZdwa6mJsxnMqgM3Cwsl7yNVqBEkCrCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d56a500576f3232f93451d0af2aa9c72cd19834268ec8b2cf78c3a18473965f7","last_reissued_at":"2026-07-05T02:14:39.063027Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:14:39.063027Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"High-Performance Large-Scale Image Recognition Without Normalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Andrew Brock, Karen Simonyan, Samuel L. Smith, Soham De","submitted_at":"2021-02-11T18:23:20Z","abstract_excerpt":"Batch normalization is a key component of most image classification models, but it has many undesirable properties stemming from its dependence on the batch size and interactions between examples. Although recent work has succeeded in training deep ResNets without normalization layers, these models do not match the test accuracies of the best batch-normalized networks, and are often unstable for large learning rates or strong data augmentations. In this work, we develop an adaptive gradient clipping technique which overcomes these instabilities, and design a significantly improved class of Nor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.06171","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.06171/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.06171","created_at":"2026-07-05T02:14:39.063088+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.06171v1","created_at":"2026-07-05T02:14:39.063088+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.06171","created_at":"2026-07-05T02:14:39.063088+00:00"},{"alias_kind":"pith_short_12","alias_value":"2VVFABLW6MRS","created_at":"2026-07-05T02:14:39.063088+00:00"},{"alias_kind":"pith_short_16","alias_value":"2VVFABLW6MRS7E2F","created_at":"2026-07-05T02:14:39.063088+00:00"},{"alias_kind":"pith_short_8","alias_value":"2VVFABLW","created_at":"2026-07-05T02:14:39.063088+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.01990","citing_title":"Deep Learning Alternatives of the Kolmogorov Superposition Theorem","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2602.10408","citing_title":"Gated Normalization Removal and Scale Anchoring in Pre-Norm Transformers","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04923","citing_title":"Imposing Boundary Conditions on Neural Operators via Learned Function Extensions","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2208.07339","citing_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL","json":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL.json","graph_json":"https://pith.science/api/pith-number/2VVFABLW6MRS7E2FDUFPFKU4OL/graph.json","events_json":"https://pith.science/api/pith-number/2VVFABLW6MRS7E2FDUFPFKU4OL/events.json","paper":"https://pith.science/paper/2VVFABLW"},"agent_actions":{"view_html":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL","download_json":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL.json","view_paper":"https://pith.science/paper/2VVFABLW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.06171&json=true","fetch_graph":"https://pith.science/api/pith-number/2VVFABLW6MRS7E2FDUFPFKU4OL/graph.json","fetch_events":"https://pith.science/api/pith-number/2VVFABLW6MRS7E2FDUFPFKU4OL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL/action/storage_attestation","attest_author":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL/action/author_attestation","sign_citation":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL/action/citation_signature","submit_replication":"https://pith.science/pith/2VVFABLW6MRS7E2FDUFPFKU4OL/action/replication_record"}},"created_at":"2026-07-05T02:14:39.063088+00:00","updated_at":"2026-07-05T02:14:39.063088+00:00"}