{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:FWD32HDPNTM7LGTXALEL6D57UB","short_pith_number":"pith:FWD32HDP","schema_version":"1.0","canonical_sha256":"2d87bd1c6f6cd9f59a7702c8bf0fbfa064df5a953e17bd812e8bd4ef694fe3b2","source":{"kind":"arxiv","id":"2006.07159","version":1},"attestation_state":"computed","paper":{"title":"Are we done with ImageNet?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"A\\\"aron van den Oord, Alexander Kolesnikov, Lucas Beyer, Olivier J. H\\'enaff, Xiaohua Zhai","submitted_at":"2020-06-12T13:17:25Z","abstract_excerpt":"Yes, and no. We ask whether recent progress on the ImageNet classification benchmark continues to represent meaningful generalization, or whether the community has started to overfit to the idiosyncrasies of its labeling procedure. We therefore develop a significantly more robust procedure for collecting human annotations of the ImageNet validation set. Using these new labels, we reassess the accuracy of recently proposed ImageNet classifiers, and find their gains to be substantially smaller than those reported on the original labels. Furthermore, we find the original ImageNet labels to no lon"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.07159","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-06-12T13:17:25Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d2a1d580f5ecb73564861955fd63a58b14e6c6cb9a7cbc4fa8bd4aaaeb804ea6","abstract_canon_sha256":"c8e074597e1933c60f7117b044e8015216e6d40dda90ffc5da2171e5e7749de0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:09:50.839591Z","signature_b64":"AOGG4xgr4aCO2d45lYOaeW55ghw6s45s+08dC7UCB6dFBJj06GjMD7ZpKDAv51ewtpkiUNPnONgsTMlT4CQaBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d87bd1c6f6cd9f59a7702c8bf0fbfa064df5a953e17bd812e8bd4ef694fe3b2","last_reissued_at":"2026-07-05T01:09:50.839173Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:09:50.839173Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are we done with ImageNet?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"A\\\"aron van den Oord, Alexander Kolesnikov, Lucas Beyer, Olivier J. H\\'enaff, Xiaohua Zhai","submitted_at":"2020-06-12T13:17:25Z","abstract_excerpt":"Yes, and no. We ask whether recent progress on the ImageNet classification benchmark continues to represent meaningful generalization, or whether the community has started to overfit to the idiosyncrasies of its labeling procedure. We therefore develop a significantly more robust procedure for collecting human annotations of the ImageNet validation set. Using these new labels, we reassess the accuracy of recently proposed ImageNet classifiers, and find their gains to be substantially smaller than those reported on the original labels. Furthermore, we find the original ImageNet labels to no lon"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.07159","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.07159/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.07159","created_at":"2026-07-05T01:09:50.839224+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.07159v1","created_at":"2026-07-05T01:09:50.839224+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.07159","created_at":"2026-07-05T01:09:50.839224+00:00"},{"alias_kind":"pith_short_12","alias_value":"FWD32HDPNTM7","created_at":"2026-07-05T01:09:50.839224+00:00"},{"alias_kind":"pith_short_16","alias_value":"FWD32HDPNTM7LGTX","created_at":"2026-07-05T01:09:50.839224+00:00"},{"alias_kind":"pith_short_8","alias_value":"FWD32HDP","created_at":"2026-07-05T01:09:50.839224+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23204","citing_title":"Unmasking LAION-5B: Age, Gender, Race, and Emotion Biases in Large-Scale Image Datasets","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13221","citing_title":"From Uncertain Judgments to Calibrated Rankings: Conformal Elo Estimation for LLM Evaluation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31397","citing_title":"Mixture-of-Control: State-Aware Fine-Tuning for Transformer-based Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2405.00892","citing_title":"Wake Vision: A Tailored Dataset and Benchmark Suite for TinyML Computer Vision Applications","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16950","citing_title":"LAION-C: An Out-of-Distribution Benchmark for Web-Scale Vision Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22570","citing_title":"CanViT: Toward Active-Vision Foundation Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15714","citing_title":"Position: Early-Stage Quality Assurance in Annotation Pipelines Is More Cost-Effective Than Late-Stage Validation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07776","citing_title":"SCOOTER: A Human Evaluation Framework for Unrestricted Adversarial Examples","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14137","citing_title":"Franca: Nested Matryoshka Clustering for Scalable Visual Representation Learning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2305.18565","citing_title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2303.15343","citing_title":"Sigmoid Loss for Language Image Pre-Training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2209.06794","citing_title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","ref_index":189,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16588","citing_title":"Vision Transformers Need Registers","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2502.14786","citing_title":"SigLIP 2: Multilingual Vision-Language Encoders with Improved Semantic Understanding, Localization, and Dense Features","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2304.07193","citing_title":"DINOv2: Learning Robust Visual Features without Supervision","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB","json":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB.json","graph_json":"https://pith.science/api/pith-number/FWD32HDPNTM7LGTXALEL6D57UB/graph.json","events_json":"https://pith.science/api/pith-number/FWD32HDPNTM7LGTXALEL6D57UB/events.json","paper":"https://pith.science/paper/FWD32HDP"},"agent_actions":{"view_html":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB","download_json":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB.json","view_paper":"https://pith.science/paper/FWD32HDP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.07159&json=true","fetch_graph":"https://pith.science/api/pith-number/FWD32HDPNTM7LGTXALEL6D57UB/graph.json","fetch_events":"https://pith.science/api/pith-number/FWD32HDPNTM7LGTXALEL6D57UB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB/action/storage_attestation","attest_author":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB/action/author_attestation","sign_citation":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB/action/citation_signature","submit_replication":"https://pith.science/pith/FWD32HDPNTM7LGTXALEL6D57UB/action/replication_record"}},"created_at":"2026-07-05T01:09:50.839224+00:00","updated_at":"2026-07-05T01:09:50.839224+00:00"}