{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NKE23PC5HDFKTEQUZS24IRAMN7","short_pith_number":"pith:NKE23PC5","schema_version":"1.0","canonical_sha256":"6a89adbc5d38caa99214ccb5c4440c6fd2af74a756feb8f4c6237407b56b6b5d","source":{"kind":"arxiv","id":"2404.00498","version":2},"attestation_state":"computed","paper":{"title":"94% on CIFAR-10 in 3.29 Seconds on a Single GPU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Keller Jordan","submitted_at":"2024-03-30T23:42:23Z","abstract_excerpt":"CIFAR-10 is among the most widely used datasets in machine learning, facilitating thousands of research projects per year. To accelerate research and reduce the cost of experiments, we introduce training methods for CIFAR-10 which reach 94% accuracy in 3.29 seconds, 95% in 10.4 seconds, and 96% in 46.3 seconds, when run on a single NVIDIA A100 GPU. As one factor contributing to these training speeds, we propose a derandomized variant of horizontal flipping augmentation, which we show improves over the standard method in every case where flipping is beneficial over no flipping at all. Our code "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00498","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-30T23:42:23Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"a63d3638aebc3cb587f11c6073123ce6c36a790917a83fb78a27f4329c6fca96","abstract_canon_sha256":"2b4b6a86a8a56b5dbd8183dbca4b576ef150cf1bc781fccd653cb1342899fe94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:39.042958Z","signature_b64":"d7kF0DwLCtXlZKtVUhaMXmOaX0tzfeMGaA6jBKwdxjDn7qmOc6t8JAVTurmmPiy90Fo2+JwNcI15RpYtMhboBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a89adbc5d38caa99214ccb5c4440c6fd2af74a756feb8f4c6237407b56b6b5d","last_reissued_at":"2026-07-05T08:04:39.042480Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:39.042480Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"94% on CIFAR-10 in 3.29 Seconds on a Single GPU","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Keller Jordan","submitted_at":"2024-03-30T23:42:23Z","abstract_excerpt":"CIFAR-10 is among the most widely used datasets in machine learning, facilitating thousands of research projects per year. To accelerate research and reduce the cost of experiments, we introduce training methods for CIFAR-10 which reach 94% accuracy in 3.29 seconds, 95% in 10.4 seconds, and 96% in 46.3 seconds, when run on a single NVIDIA A100 GPU. As one factor contributing to these training speeds, we propose a derandomized variant of horizontal flipping augmentation, which we show improves over the standard method in every case where flipping is beneficial over no flipping at all. Our code "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00498","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00498/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00498","created_at":"2026-07-05T08:04:39.042538+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00498v2","created_at":"2026-07-05T08:04:39.042538+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00498","created_at":"2026-07-05T08:04:39.042538+00:00"},{"alias_kind":"pith_short_12","alias_value":"NKE23PC5HDFK","created_at":"2026-07-05T08:04:39.042538+00:00"},{"alias_kind":"pith_short_16","alias_value":"NKE23PC5HDFKTEQU","created_at":"2026-07-05T08:04:39.042538+00:00"},{"alias_kind":"pith_short_8","alias_value":"NKE23PC5","created_at":"2026-07-05T08:04:39.042538+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18528","citing_title":"Scale-Invariant Neural Network Optimization: Norm Geometry and Heavy-Tailed Noise","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18528","citing_title":"Scale-Invariant Neural Network Optimization: Norm Geometry and Heavy-Tailed Noise","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11838","citing_title":"Gradient Clipping Beyond Vector Norms: A Spectral Approach for Matrix-Valued Parameters","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06884","citing_title":"Muon with Nesterov Momentum: Heavy-Tailed Noise and (Randomized) Inexact Polar Decomposition","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7","json":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7.json","graph_json":"https://pith.science/api/pith-number/NKE23PC5HDFKTEQUZS24IRAMN7/graph.json","events_json":"https://pith.science/api/pith-number/NKE23PC5HDFKTEQUZS24IRAMN7/events.json","paper":"https://pith.science/paper/NKE23PC5"},"agent_actions":{"view_html":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7","download_json":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7.json","view_paper":"https://pith.science/paper/NKE23PC5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00498&json=true","fetch_graph":"https://pith.science/api/pith-number/NKE23PC5HDFKTEQUZS24IRAMN7/graph.json","fetch_events":"https://pith.science/api/pith-number/NKE23PC5HDFKTEQUZS24IRAMN7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7/action/storage_attestation","attest_author":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7/action/author_attestation","sign_citation":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7/action/citation_signature","submit_replication":"https://pith.science/pith/NKE23PC5HDFKTEQUZS24IRAMN7/action/replication_record"}},"created_at":"2026-07-05T08:04:39.042538+00:00","updated_at":"2026-07-05T08:04:39.042538+00:00"}