{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GPXRIQ37JXQZHIIF3X4PBOQBCQ","short_pith_number":"pith:GPXRIQ37","schema_version":"1.0","canonical_sha256":"33ef14437f4de193a105ddf8f0ba011416df277f5b2c2b20c156bf3f8b578ebd","source":{"kind":"arxiv","id":"2108.02922","version":2},"attestation_state":"computed","paper":{"title":"Mitigating Dataset Harms Requires Stewardship: Lessons from 1000 Papers","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.LG","authors_text":"Arunesh Mathur, Arvind Narayanan, Kenny Peng","submitted_at":"2021-08-06T02:52:36Z","abstract_excerpt":"Machine learning datasets have elicited concerns about privacy, bias, and unethical applications, leading to the retraction of prominent datasets such as DukeMTMC, MS-Celeb-1M, and Tiny Images. In response, the machine learning community has called for higher ethical standards in dataset creation. To help inform these efforts, we studied three influential but ethically problematic face and person recognition datasets -- Labeled Faces in the Wild (LFW), MS-Celeb-1M, and DukeMTM -- by analyzing nearly 1000 papers that cite them. We found that the creation of derivative datasets and models, broad"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.02922","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2021-08-06T02:52:36Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"f0a3f4c55de45ce314dea33b4f6e0e7c61b761497c6ae52e8d722e0b39032550","abstract_canon_sha256":"f5a64a2b36007f19f5f15c36dfbb27134c08b5be6ec90d20f0a74243ce5a4bce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:33:46.585446Z","signature_b64":"cxg9JJcenwT8zRAkO9w18f/dQq3CsW0666uW+g7hgr9LS1eERDXoYxc+SGfIRThcfeQ6sBrLzt+d0887Y2RiBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33ef14437f4de193a105ddf8f0ba011416df277f5b2c2b20c156bf3f8b578ebd","last_reissued_at":"2026-07-05T03:33:46.585087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:33:46.585087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Dataset Harms Requires Stewardship: Lessons from 1000 Papers","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.LG","authors_text":"Arunesh Mathur, Arvind Narayanan, Kenny Peng","submitted_at":"2021-08-06T02:52:36Z","abstract_excerpt":"Machine learning datasets have elicited concerns about privacy, bias, and unethical applications, leading to the retraction of prominent datasets such as DukeMTMC, MS-Celeb-1M, and Tiny Images. In response, the machine learning community has called for higher ethical standards in dataset creation. To help inform these efforts, we studied three influential but ethically problematic face and person recognition datasets -- Labeled Faces in the Wild (LFW), MS-Celeb-1M, and DukeMTM -- by analyzing nearly 1000 papers that cite them. We found that the creation of derivative datasets and models, broad"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.02922","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.02922/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.02922","created_at":"2026-07-05T03:33:46.585139+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.02922v2","created_at":"2026-07-05T03:33:46.585139+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.02922","created_at":"2026-07-05T03:33:46.585139+00:00"},{"alias_kind":"pith_short_12","alias_value":"GPXRIQ37JXQZ","created_at":"2026-07-05T03:33:46.585139+00:00"},{"alias_kind":"pith_short_16","alias_value":"GPXRIQ37JXQZHIIF","created_at":"2026-07-05T03:33:46.585139+00:00"},{"alias_kind":"pith_short_8","alias_value":"GPXRIQ37","created_at":"2026-07-05T03:33:46.585139+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.00106","citing_title":"LicenseGPT: A Fine-tuned Foundation Model for Publicly Available Dataset License Compliance","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ","json":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ.json","graph_json":"https://pith.science/api/pith-number/GPXRIQ37JXQZHIIF3X4PBOQBCQ/graph.json","events_json":"https://pith.science/api/pith-number/GPXRIQ37JXQZHIIF3X4PBOQBCQ/events.json","paper":"https://pith.science/paper/GPXRIQ37"},"agent_actions":{"view_html":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ","download_json":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ.json","view_paper":"https://pith.science/paper/GPXRIQ37","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.02922&json=true","fetch_graph":"https://pith.science/api/pith-number/GPXRIQ37JXQZHIIF3X4PBOQBCQ/graph.json","fetch_events":"https://pith.science/api/pith-number/GPXRIQ37JXQZHIIF3X4PBOQBCQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ/action/storage_attestation","attest_author":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ/action/author_attestation","sign_citation":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ/action/citation_signature","submit_replication":"https://pith.science/pith/GPXRIQ37JXQZHIIF3X4PBOQBCQ/action/replication_record"}},"created_at":"2026-07-05T03:33:46.585139+00:00","updated_at":"2026-07-05T03:33:46.585139+00:00"}