{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SN7WVNWUUOTDICTPFAAAH6E36H","short_pith_number":"pith:SN7WVNWU","schema_version":"1.0","canonical_sha256":"937f6ab6d4a3a6340a6f280003f89bf1ca8ee7673ecdbf2a931619f1e2e8ce44","source":{"kind":"arxiv","id":"2310.17589","version":3},"attestation_state":"computed","paper":{"title":"An Open Source Data Contamination Report for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenghua Lin, Frank Guerin, Yucheng Li","submitted_at":"2023-10-26T17:11:42Z","abstract_excerpt":"Data contamination in model evaluation has become increasingly prevalent with the growing popularity of large language models. It allows models to \"cheat\" via memorisation instead of displaying true capabilities. Therefore, contamination analysis has become an crucial part of reliable model evaluation to validate results. However, existing contamination analysis is usually conducted internally by large language model developers and often lacks transparency and completeness. This paper presents an extensive data contamination report for over 15 popular large language models across six popular m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17589","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-26T17:11:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3ae520b3e936190ccbf9e4a391c81d07b054035457d7bcb1bf82cd3e048c0b52","abstract_canon_sha256":"6f1edf977f470e0770a38da64ce2be7601c8ad0dff1f7ce4f655e9cade5d2257"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:38:32.141815Z","signature_b64":"YQif8a6FFz7+UA1wA+h2t+supsi4m5ka4Y6LtQ/87/qGH9Qr5UOb2NtqAVb4uHehaYzMQxA2ZFleya+EoUxNDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"937f6ab6d4a3a6340a6f280003f89bf1ca8ee7673ecdbf2a931619f1e2e8ce44","last_reissued_at":"2026-07-05T07:38:32.141360Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:38:32.141360Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Open Source Data Contamination Report for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenghua Lin, Frank Guerin, Yucheng Li","submitted_at":"2023-10-26T17:11:42Z","abstract_excerpt":"Data contamination in model evaluation has become increasingly prevalent with the growing popularity of large language models. It allows models to \"cheat\" via memorisation instead of displaying true capabilities. Therefore, contamination analysis has become an crucial part of reliable model evaluation to validate results. However, existing contamination analysis is usually conducted internally by large language model developers and often lacks transparency and completeness. This paper presents an extensive data contamination report for over 15 popular large language models across six popular m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17589","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17589/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17589","created_at":"2026-07-05T07:38:32.141426+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17589v3","created_at":"2026-07-05T07:38:32.141426+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17589","created_at":"2026-07-05T07:38:32.141426+00:00"},{"alias_kind":"pith_short_12","alias_value":"SN7WVNWUUOTD","created_at":"2026-07-05T07:38:32.141426+00:00"},{"alias_kind":"pith_short_16","alias_value":"SN7WVNWUUOTDICTP","created_at":"2026-07-05T07:38:32.141426+00:00"},{"alias_kind":"pith_short_8","alias_value":"SN7WVNWU","created_at":"2026-07-05T07:38:32.141426+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04244","citing_title":"Benchmark Data Contamination of Large Language Models: A Survey","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09275","citing_title":"Inflated Excellence or True Performance? Rethinking Medical Diagnostic Benchmarks with Dynamic Evaluation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17771","citing_title":"SPENCE: A Syntactic Probe for Detecting Contamination in NL2SQL Benchmarks","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H","json":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H.json","graph_json":"https://pith.science/api/pith-number/SN7WVNWUUOTDICTPFAAAH6E36H/graph.json","events_json":"https://pith.science/api/pith-number/SN7WVNWUUOTDICTPFAAAH6E36H/events.json","paper":"https://pith.science/paper/SN7WVNWU"},"agent_actions":{"view_html":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H","download_json":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H.json","view_paper":"https://pith.science/paper/SN7WVNWU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17589&json=true","fetch_graph":"https://pith.science/api/pith-number/SN7WVNWUUOTDICTPFAAAH6E36H/graph.json","fetch_events":"https://pith.science/api/pith-number/SN7WVNWUUOTDICTPFAAAH6E36H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H/action/storage_attestation","attest_author":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H/action/author_attestation","sign_citation":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H/action/citation_signature","submit_replication":"https://pith.science/pith/SN7WVNWUUOTDICTPFAAAH6E36H/action/replication_record"}},"created_at":"2026-07-05T07:38:32.141426+00:00","updated_at":"2026-07-05T07:38:32.141426+00:00"}