{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WML3K5MZ5GDESO7MOVAIGXMUZ6","short_pith_number":"pith:WML3K5MZ","schema_version":"1.0","canonical_sha256":"b317b57599e986493bec7540835d94cf9ce17a3078f0c0be1f57c9e3f4c8aa61","source":{"kind":"arxiv","id":"2401.06059","version":1},"attestation_state":"computed","paper":{"title":"Investigating Data Contamination for Pre-training Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiawei Han, Ken Ziyu Liu, Ming Zhong, Minhao Jiang, Rylan Schaeffer, Sanmi Koyejo, Siru Ouyang","submitted_at":"2024-01-11T17:24:49Z","abstract_excerpt":"Language models pre-trained on web-scale corpora demonstrate impressive capabilities on diverse downstream tasks. However, there is increasing concern whether such capabilities might arise from evaluation datasets being included in the pre-training corpus -- a phenomenon known as \\textit{data contamination} -- in a manner that artificially increases performance. There has been little understanding of how this potential contamination might influence LMs' performance on downstream tasks. In this paper, we explore the impact of data contamination at the pre-training stage by pre-training a series"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06059","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-11T17:24:49Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3f683ebc6f8c0d03bb377ab7be1f7556278e7cb68a3fddd14da323fada976858","abstract_canon_sha256":"dbdfdf732b900702c1009d57ed770ac938d07d83bc2a513991f080f75fcc95e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:32:40.120830Z","signature_b64":"SdcvdGwU0Ne0t1NEoCZDqbQktHdSNpke7duLlxCpVc7/q7sJ40rUtI6ZX0zKF1qjBkuwOB6V2132f64rci6eAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b317b57599e986493bec7540835d94cf9ce17a3078f0c0be1f57c9e3f4c8aa61","last_reissued_at":"2026-07-05T07:32:40.120322Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:32:40.120322Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating Data Contamination for Pre-training Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiawei Han, Ken Ziyu Liu, Ming Zhong, Minhao Jiang, Rylan Schaeffer, Sanmi Koyejo, Siru Ouyang","submitted_at":"2024-01-11T17:24:49Z","abstract_excerpt":"Language models pre-trained on web-scale corpora demonstrate impressive capabilities on diverse downstream tasks. However, there is increasing concern whether such capabilities might arise from evaluation datasets being included in the pre-training corpus -- a phenomenon known as \\textit{data contamination} -- in a manner that artificially increases performance. There has been little understanding of how this potential contamination might influence LMs' performance on downstream tasks. In this paper, we explore the impact of data contamination at the pre-training stage by pre-training a series"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06059","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06059/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06059","created_at":"2026-07-05T07:32:40.120388+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06059v1","created_at":"2026-07-05T07:32:40.120388+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06059","created_at":"2026-07-05T07:32:40.120388+00:00"},{"alias_kind":"pith_short_12","alias_value":"WML3K5MZ5GDE","created_at":"2026-07-05T07:32:40.120388+00:00"},{"alias_kind":"pith_short_16","alias_value":"WML3K5MZ5GDESO7M","created_at":"2026-07-05T07:32:40.120388+00:00"},{"alias_kind":"pith_short_8","alias_value":"WML3K5MZ","created_at":"2026-07-05T07:32:40.120388+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26161","citing_title":"TSFMAudit: Data Contamination Auditing in Forecasting Time Series Foundation Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19999","citing_title":"LLM Benchmark Datasets Should Be Contamination-Resistant","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06865","citing_title":"Dataset Watermarking for Closed LLMs with Provable Detection","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6","json":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6.json","graph_json":"https://pith.science/api/pith-number/WML3K5MZ5GDESO7MOVAIGXMUZ6/graph.json","events_json":"https://pith.science/api/pith-number/WML3K5MZ5GDESO7MOVAIGXMUZ6/events.json","paper":"https://pith.science/paper/WML3K5MZ"},"agent_actions":{"view_html":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6","download_json":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6.json","view_paper":"https://pith.science/paper/WML3K5MZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06059&json=true","fetch_graph":"https://pith.science/api/pith-number/WML3K5MZ5GDESO7MOVAIGXMUZ6/graph.json","fetch_events":"https://pith.science/api/pith-number/WML3K5MZ5GDESO7MOVAIGXMUZ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6/action/storage_attestation","attest_author":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6/action/author_attestation","sign_citation":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6/action/citation_signature","submit_replication":"https://pith.science/pith/WML3K5MZ5GDESO7MOVAIGXMUZ6/action/replication_record"}},"created_at":"2026-07-05T07:32:40.120388+00:00","updated_at":"2026-07-05T07:32:40.120388+00:00"}