{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H5DXSS2V3SIEHUBEFUSQB3DLSH","short_pith_number":"pith:H5DXSS2V","schema_version":"1.0","canonical_sha256":"3f47794b55dc9043d0242d2500ec6b91d86288cda3e5c8f77f48d40d9528afc4","source":{"kind":"arxiv","id":"2412.18409","version":2},"attestation_state":"computed","paper":{"title":"The Impact of the Single-Label Assumption in Image Recognition Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Arnout Van Messem, Esla Timothy Anzaku, Seyed Amir Mousavi, Wesley De Neve","submitted_at":"2024-12-24T12:55:31Z","abstract_excerpt":"Deep neural networks (DNNs) are typically evaluated under the assumption that each image has a single correct label. However, many images in benchmarks like ImageNet contain multiple valid labels, creating a mismatch between evaluation protocols and the actual complexity of visual data. This mismatch can penalize DNNs for predicting correct but unannotated labels, which may partly explain reported accuracy drops, such as the widely cited 11 to 14 percent top-1 accuracy decline on ImageNetV2, a replication test set for ImageNet. This raises the question: do such drops reflect genuine generaliza"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18409","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-24T12:55:31Z","cross_cats_sorted":[],"title_canon_sha256":"fe8a256035c1dad95064f60cc9fa1ae4ac8ee5f7abcb52baa345915fb90d7325","abstract_canon_sha256":"065405e1198d691ad24e0339c288519c26d1c05c696ca9594a312c7f2222ab7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:49.916102Z","signature_b64":"Liwa0Y0HvmWEfh3I2bD3/CpCroWCPI2b6LxkfkjSZi+0pUMaw/sBT8NnpqdlSnPqbGUfX5JP+TIiRGK/hN33Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f47794b55dc9043d0242d2500ec6b91d86288cda3e5c8f77f48d40d9528afc4","last_reissued_at":"2026-07-05T11:10:49.915606Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:49.915606Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Impact of the Single-Label Assumption in Image Recognition Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Arnout Van Messem, Esla Timothy Anzaku, Seyed Amir Mousavi, Wesley De Neve","submitted_at":"2024-12-24T12:55:31Z","abstract_excerpt":"Deep neural networks (DNNs) are typically evaluated under the assumption that each image has a single correct label. However, many images in benchmarks like ImageNet contain multiple valid labels, creating a mismatch between evaluation protocols and the actual complexity of visual data. This mismatch can penalize DNNs for predicting correct but unannotated labels, which may partly explain reported accuracy drops, such as the widely cited 11 to 14 percent top-1 accuracy decline on ImageNetV2, a replication test set for ImageNet. This raises the question: do such drops reflect genuine generaliza"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18409","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18409/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18409","created_at":"2026-07-05T11:10:49.915659+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18409v2","created_at":"2026-07-05T11:10:49.915659+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18409","created_at":"2026-07-05T11:10:49.915659+00:00"},{"alias_kind":"pith_short_12","alias_value":"H5DXSS2V3SIE","created_at":"2026-07-05T11:10:49.915659+00:00"},{"alias_kind":"pith_short_16","alias_value":"H5DXSS2V3SIEHUBE","created_at":"2026-07-05T11:10:49.915659+00:00"},{"alias_kind":"pith_short_8","alias_value":"H5DXSS2V","created_at":"2026-07-05T11:10:49.915659+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH","json":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH.json","graph_json":"https://pith.science/api/pith-number/H5DXSS2V3SIEHUBEFUSQB3DLSH/graph.json","events_json":"https://pith.science/api/pith-number/H5DXSS2V3SIEHUBEFUSQB3DLSH/events.json","paper":"https://pith.science/paper/H5DXSS2V"},"agent_actions":{"view_html":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH","download_json":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH.json","view_paper":"https://pith.science/paper/H5DXSS2V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18409&json=true","fetch_graph":"https://pith.science/api/pith-number/H5DXSS2V3SIEHUBEFUSQB3DLSH/graph.json","fetch_events":"https://pith.science/api/pith-number/H5DXSS2V3SIEHUBEFUSQB3DLSH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH/action/storage_attestation","attest_author":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH/action/author_attestation","sign_citation":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH/action/citation_signature","submit_replication":"https://pith.science/pith/H5DXSS2V3SIEHUBEFUSQB3DLSH/action/replication_record"}},"created_at":"2026-07-05T11:10:49.915659+00:00","updated_at":"2026-07-05T11:10:49.915659+00:00"}