{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:O6342L5MBOOITQVTOFSCWACFCF","short_pith_number":"pith:O6342L5M","schema_version":"1.0","canonical_sha256":"77b7cd2fac0b9c89c2b371642b0045116a6b4699b5a79d5479d0f3d41483af49","source":{"kind":"arxiv","id":"2111.04424","version":2},"attestation_state":"computed","paper":{"title":"A Framework for Deprecating Datasets: Standardizing Documentation, Identification, and Communication","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CY","authors_text":"Alexandra Sasha Luccioni, Frances Corry, Hamsini Sridharan, Jason Schultz, Kate Crawford, Mike Ananny","submitted_at":"2021-10-18T20:13:51Z","abstract_excerpt":"Datasets are central to training machine learning (ML) models. The ML community has recently made significant improvements to data stewardship and documentation practices across the model development life cycle. However, the act of deprecating, or deleting, datasets has been largely overlooked, and there are currently no standardized approaches for structuring this stage of the dataset life cycle. In this paper, we study the practice of dataset deprecation in ML, identify several cases of datasets that continued to circulate despite having been deprecated, and describe the different technical,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.04424","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CY","submitted_at":"2021-10-18T20:13:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"23af8cfa1d69a7375c69cee26ea002926f947a62f6d8a6bfb53382c7ab717948","abstract_canon_sha256":"970c87bb97bd54af3b802e76a2e9e146560da4eb67ce8b08cb7b8b356abd129e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:21:26.476340Z","signature_b64":"iB1eWMImXQtwbJitRZ51bua6AtILPgG/upOlT5Do/zkX+234+esAZM0N6pK1+aF/x8/k4OJ4r6AYSaNI+1KIBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77b7cd2fac0b9c89c2b371642b0045116a6b4699b5a79d5479d0f3d41483af49","last_reissued_at":"2026-07-05T04:21:26.475835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:21:26.475835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Framework for Deprecating Datasets: Standardizing Documentation, Identification, and Communication","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CY","authors_text":"Alexandra Sasha Luccioni, Frances Corry, Hamsini Sridharan, Jason Schultz, Kate Crawford, Mike Ananny","submitted_at":"2021-10-18T20:13:51Z","abstract_excerpt":"Datasets are central to training machine learning (ML) models. The ML community has recently made significant improvements to data stewardship and documentation practices across the model development life cycle. However, the act of deprecating, or deleting, datasets has been largely overlooked, and there are currently no standardized approaches for structuring this stage of the dataset life cycle. In this paper, we study the practice of dataset deprecation in ML, identify several cases of datasets that continued to circulate despite having been deprecated, and describe the different technical,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.04424","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.04424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.04424","created_at":"2026-07-05T04:21:26.475898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.04424v2","created_at":"2026-07-05T04:21:26.475898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.04424","created_at":"2026-07-05T04:21:26.475898+00:00"},{"alias_kind":"pith_short_12","alias_value":"O6342L5MBOOI","created_at":"2026-07-05T04:21:26.475898+00:00"},{"alias_kind":"pith_short_16","alias_value":"O6342L5MBOOITQVT","created_at":"2026-07-05T04:21:26.475898+00:00"},{"alias_kind":"pith_short_8","alias_value":"O6342L5M","created_at":"2026-07-05T04:21:26.475898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2503.13463","citing_title":"Completeness of Datasets Documentation on ML/AI repositories: an Empirical Investigation","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF","json":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF.json","graph_json":"https://pith.science/api/pith-number/O6342L5MBOOITQVTOFSCWACFCF/graph.json","events_json":"https://pith.science/api/pith-number/O6342L5MBOOITQVTOFSCWACFCF/events.json","paper":"https://pith.science/paper/O6342L5M"},"agent_actions":{"view_html":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF","download_json":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF.json","view_paper":"https://pith.science/paper/O6342L5M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.04424&json=true","fetch_graph":"https://pith.science/api/pith-number/O6342L5MBOOITQVTOFSCWACFCF/graph.json","fetch_events":"https://pith.science/api/pith-number/O6342L5MBOOITQVTOFSCWACFCF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF/action/storage_attestation","attest_author":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF/action/author_attestation","sign_citation":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF/action/citation_signature","submit_replication":"https://pith.science/pith/O6342L5MBOOITQVTOFSCWACFCF/action/replication_record"}},"created_at":"2026-07-05T04:21:26.475898+00:00","updated_at":"2026-07-05T04:21:26.475898+00:00"}