{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ASEKCBTDOQZKVX2FPOB6IR6NEH","short_pith_number":"pith:ASEKCBTD","schema_version":"1.0","canonical_sha256":"0488a106637432aadf457b83e447cd21f01b074bde442b645d6eaf5fd9c81988","source":{"kind":"arxiv","id":"2403.08632","version":2},"attestation_state":"computed","paper":{"title":"A Decade's Battle on Dataset Bias: Are We There Yet?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Kaiming He, Zhuang Liu","submitted_at":"2024-03-13T15:46:37Z","abstract_excerpt":"We revisit the \"dataset classification\" experiment suggested by Torralba & Efros (2011) a decade ago, in the new era with large-scale, diverse, and hopefully less biased datasets as well as more capable neural network architectures. Surprisingly, we observe that modern neural networks can achieve excellent accuracy in classifying which dataset an image is from: e.g., we report 84.7% accuracy on held-out validation data for the three-way classification problem consisting of the YFCC, CC, and DataComp datasets. Our further experiments show that such a dataset classifier could learn semantic feat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08632","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-13T15:46:37Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e81c8f84d35bfab99db244171b0eba26c22c1a9ba26f56ee82b5e95a7a437d71","abstract_canon_sha256":"7b4e4b0b225ac85e2bf0146e77411c07fcaeb2d4422cbc8bed55b1591a45b3fb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:44.788067Z","signature_b64":"Ws0wrXeorCdXVQtkVEiuLONn6RedDo3enoCOZfhurLSKA7tnBXr/MVQgqEdqrZQpUHg6B6cK2B2HAiRrTg9yCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0488a106637432aadf457b83e447cd21f01b074bde442b645d6eaf5fd9c81988","last_reissued_at":"2026-07-05T10:22:44.787330Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:44.787330Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Decade's Battle on Dataset Bias: Are We There Yet?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Kaiming He, Zhuang Liu","submitted_at":"2024-03-13T15:46:37Z","abstract_excerpt":"We revisit the \"dataset classification\" experiment suggested by Torralba & Efros (2011) a decade ago, in the new era with large-scale, diverse, and hopefully less biased datasets as well as more capable neural network architectures. Surprisingly, we observe that modern neural networks can achieve excellent accuracy in classifying which dataset an image is from: e.g., we report 84.7% accuracy on held-out validation data for the three-way classification problem consisting of the YFCC, CC, and DataComp datasets. Our further experiments show that such a dataset classifier could learn semantic feat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08632","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08632","created_at":"2026-07-05T10:22:44.787428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08632v2","created_at":"2026-07-05T10:22:44.787428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08632","created_at":"2026-07-05T10:22:44.787428+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASEKCBTDOQZK","created_at":"2026-07-05T10:22:44.787428+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASEKCBTDOQZKVX2F","created_at":"2026-07-05T10:22:44.787428+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASEKCBTD","created_at":"2026-07-05T10:22:44.787428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.17086","citing_title":"Advancing Multi-Agent RAG Systems with Minimalist Reinforcement Learning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2406.16860","citing_title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11237","citing_title":"DeconDTN-Toolkit: A Library for Evaluation and Enhancement of Robustness to Provenance Shift","ref_index":116,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH","json":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH.json","graph_json":"https://pith.science/api/pith-number/ASEKCBTDOQZKVX2FPOB6IR6NEH/graph.json","events_json":"https://pith.science/api/pith-number/ASEKCBTDOQZKVX2FPOB6IR6NEH/events.json","paper":"https://pith.science/paper/ASEKCBTD"},"agent_actions":{"view_html":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH","download_json":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH.json","view_paper":"https://pith.science/paper/ASEKCBTD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08632&json=true","fetch_graph":"https://pith.science/api/pith-number/ASEKCBTDOQZKVX2FPOB6IR6NEH/graph.json","fetch_events":"https://pith.science/api/pith-number/ASEKCBTDOQZKVX2FPOB6IR6NEH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH/action/storage_attestation","attest_author":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH/action/author_attestation","sign_citation":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH/action/citation_signature","submit_replication":"https://pith.science/pith/ASEKCBTDOQZKVX2FPOB6IR6NEH/action/replication_record"}},"created_at":"2026-07-05T10:22:44.787428+00:00","updated_at":"2026-07-05T10:22:44.787428+00:00"}