{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6DLC3FX56FZCK7PUY2ZTU2HV3I","short_pith_number":"pith:6DLC3FX5","schema_version":"1.0","canonical_sha256":"f0d62d96fdf172257df4c6b33a68f5da188a34304906b8a75668ef6be975f805","source":{"kind":"arxiv","id":"2310.16787","version":3},"attestation_state":"computed","paper":{"title":"The Data Provenance Initiative: A Large Scale Audit of Dataset Licensing & Attribution in AI","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anthony Chen, Damien Sileo, Enrico Shippole, Jad Kabbara, Kartik Perisetla, Kurt Bollacker, Luis Villa, Naana Obeng-Marnu, Nathan Khazam, Niklas Muennighoff, Robert Mahari, Sandy Pentland, Sara Hooker, Shayne Longpre, Tongshuang Wu, William Brannon, Xinyi Wu","submitted_at":"2023-10-25T17:20:26Z","abstract_excerpt":"The race to train language models on vast, diverse, and inconsistently documented datasets has raised pressing concerns about the legal and ethical risks for practitioners. To remedy these practices threatening data transparency and understanding, we convene a multi-disciplinary effort between legal and machine learning experts to systematically audit and trace 1800+ text datasets. We develop tools and standards to trace the lineage of these datasets, from their source, creators, series of license conditions, properties, and subsequent use. Our landscape analysis highlights the sharp divides i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.16787","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-25T17:20:26Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"1897d095e2f223221de6eeee4970116dd55d7932fb823453b88441d6257c2644","abstract_canon_sha256":"7a99f3f345ae86e5fc67f0af742a13827bf9478764929cf64891106bba17b963"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:03.167536Z","signature_b64":"0+h/e1lpARKSnhypDqYZ2201x6xHyDCFPm9qHi0vznkrBp/XJQV/fP0Ve8vIpg5w9k2pZGYu3CYM1fEFK3HoCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f0d62d96fdf172257df4c6b33a68f5da188a34304906b8a75668ef6be975f805","last_reissued_at":"2026-07-05T07:09:03.167090Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:03.167090Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Data Provenance Initiative: A Large Scale Audit of Dataset Licensing & Attribution in AI","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anthony Chen, Damien Sileo, Enrico Shippole, Jad Kabbara, Kartik Perisetla, Kurt Bollacker, Luis Villa, Naana Obeng-Marnu, Nathan Khazam, Niklas Muennighoff, Robert Mahari, Sandy Pentland, Sara Hooker, Shayne Longpre, Tongshuang Wu, William Brannon, Xinyi Wu","submitted_at":"2023-10-25T17:20:26Z","abstract_excerpt":"The race to train language models on vast, diverse, and inconsistently documented datasets has raised pressing concerns about the legal and ethical risks for practitioners. To remedy these practices threatening data transparency and understanding, we convene a multi-disciplinary effort between legal and machine learning experts to systematically audit and trace 1800+ text datasets. We develop tools and standards to trace the lineage of these datasets, from their source, creators, series of license conditions, properties, and subsequent use. Our landscape analysis highlights the sharp divides i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16787","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16787/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.16787","created_at":"2026-07-05T07:09:03.167154+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.16787v3","created_at":"2026-07-05T07:09:03.167154+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16787","created_at":"2026-07-05T07:09:03.167154+00:00"},{"alias_kind":"pith_short_12","alias_value":"6DLC3FX56FZC","created_at":"2026-07-05T07:09:03.167154+00:00"},{"alias_kind":"pith_short_16","alias_value":"6DLC3FX56FZCK7PU","created_at":"2026-07-05T07:09:03.167154+00:00"},{"alias_kind":"pith_short_8","alias_value":"6DLC3FX5","created_at":"2026-07-05T07:09:03.167154+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29464","citing_title":"Rank-Aware Hyperbolic Alignment for Vision-Language Dataset Distillation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00106","citing_title":"LicenseGPT: A Fine-tuned Foundation Model for Publicly Available Dataset License Compliance","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19173","citing_title":"StarCoder 2 and The Stack v2: The Next Generation","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07190","citing_title":"The ATOM Report: Measuring the Open Language Model Ecosystem","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I","json":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I.json","graph_json":"https://pith.science/api/pith-number/6DLC3FX56FZCK7PUY2ZTU2HV3I/graph.json","events_json":"https://pith.science/api/pith-number/6DLC3FX56FZCK7PUY2ZTU2HV3I/events.json","paper":"https://pith.science/paper/6DLC3FX5"},"agent_actions":{"view_html":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I","download_json":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I.json","view_paper":"https://pith.science/paper/6DLC3FX5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.16787&json=true","fetch_graph":"https://pith.science/api/pith-number/6DLC3FX56FZCK7PUY2ZTU2HV3I/graph.json","fetch_events":"https://pith.science/api/pith-number/6DLC3FX56FZCK7PUY2ZTU2HV3I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I/action/storage_attestation","attest_author":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I/action/author_attestation","sign_citation":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I/action/citation_signature","submit_replication":"https://pith.science/pith/6DLC3FX56FZCK7PUY2ZTU2HV3I/action/replication_record"}},"created_at":"2026-07-05T07:09:03.167154+00:00","updated_at":"2026-07-05T07:09:03.167154+00:00"}