{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K6C2GOHFAYEVXL7M7B2KOYCKJ5","short_pith_number":"pith:K6C2GOHF","schema_version":"1.0","canonical_sha256":"5785a338e506095bafecf874a7604a4f5c3443fcd4af2326ca695b2c351a473d","source":{"kind":"arxiv","id":"2502.14425","version":2},"attestation_state":"computed","paper":{"title":"A Survey on Data Contamination for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Yi Chang, Yuan Wu, Yuxing Cheng","submitted_at":"2025-02-20T10:23:27Z","abstract_excerpt":"Recent advancements in Large Language Models (LLMs) have demonstrated significant progress in various areas, such as text generation and code synthesis. However, the reliability of performance evaluation has come under scrutiny due to data contamination-the unintended overlap between training and test datasets. This overlap has the potential to artificially inflate model performance, as LLMs are typically trained on extensive datasets scraped from publicly available sources. These datasets often inadvertently overlap with the benchmarks used for evaluation, leading to an overestimation of the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14425","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-20T10:23:27Z","cross_cats_sorted":[],"title_canon_sha256":"8f0da2076a1d1455b77ebd7f471605f7741d41518c2e3cd4215b090f176912e9","abstract_canon_sha256":"49753bbac55a7aa86076b3c0e13b1a4005461d39627bcfbd553662f408ff1517"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:06.420565Z","signature_b64":"YeOLA2v47X0P7bwhbMCCaYoXJVrxsruHdW0UztrxzWzN9ZW4iCtKVD+wvGwMUiu2OVerLTAfPv/X8jtN2xZaDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5785a338e506095bafecf874a7604a4f5c3443fcd4af2326ca695b2c351a473d","last_reissued_at":"2026-07-05T11:16:06.420004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:06.420004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Data Contamination for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Yi Chang, Yuan Wu, Yuxing Cheng","submitted_at":"2025-02-20T10:23:27Z","abstract_excerpt":"Recent advancements in Large Language Models (LLMs) have demonstrated significant progress in various areas, such as text generation and code synthesis. However, the reliability of performance evaluation has come under scrutiny due to data contamination-the unintended overlap between training and test datasets. This overlap has the potential to artificially inflate model performance, as LLMs are typically trained on extensive datasets scraped from publicly available sources. These datasets often inadvertently overlap with the benchmarks used for evaluation, leading to an overestimation of the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14425","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14425","created_at":"2026-07-05T11:16:06.420073+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14425v2","created_at":"2026-07-05T11:16:06.420073+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14425","created_at":"2026-07-05T11:16:06.420073+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6C2GOHFAYEV","created_at":"2026-07-05T11:16:06.420073+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6C2GOHFAYEVXL7M","created_at":"2026-07-05T11:16:06.420073+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6C2GOHF","created_at":"2026-07-05T11:16:06.420073+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20950","citing_title":"Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20950","citing_title":"Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18284","citing_title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09395","citing_title":"Empirical Study for Structured Output Control in LLMs for Software Engineering","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30524","citing_title":"The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00053","citing_title":"SWE-Router: Routing in Multi-turn Agentic Software Engineering Tasks","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24079","citing_title":"TRACER: A Semantic-Aware Framework for Fine-Grained Contamination Detection in Code LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30524","citing_title":"The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26971","citing_title":"RLVR Datasets and Where to Find Them: Tracing Data Lineage for Better Training Data","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17181","citing_title":"A Study of LLMs' Preferences for Libraries and Programming Languages","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21856","citing_title":"The Illusion of Reasoning: Exposing Evasive Data Contamination in LLMs via Zero-CoT Truncation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21543","citing_title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22359","citing_title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14164","citing_title":"Unsteady Metrics and Benchmarking Cultures of AI Model Builders","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16392","citing_title":"RoMathExam: A Longitudinal Dataset of Romanian Math Exams (1895-2025) with a Seven-Decade Core (1957-2025)","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16941","citing_title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18543","citing_title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04325","citing_title":"Beyond the Leaderboard: Rethinking Medical Benchmarks for Large Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02442","citing_title":"Measuring AI Reasoning: A Guide for Researchers","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5","json":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5.json","graph_json":"https://pith.science/api/pith-number/K6C2GOHFAYEVXL7M7B2KOYCKJ5/graph.json","events_json":"https://pith.science/api/pith-number/K6C2GOHFAYEVXL7M7B2KOYCKJ5/events.json","paper":"https://pith.science/paper/K6C2GOHF"},"agent_actions":{"view_html":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5","download_json":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5.json","view_paper":"https://pith.science/paper/K6C2GOHF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14425&json=true","fetch_graph":"https://pith.science/api/pith-number/K6C2GOHFAYEVXL7M7B2KOYCKJ5/graph.json","fetch_events":"https://pith.science/api/pith-number/K6C2GOHFAYEVXL7M7B2KOYCKJ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5/action/storage_attestation","attest_author":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5/action/author_attestation","sign_citation":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5/action/citation_signature","submit_replication":"https://pith.science/pith/K6C2GOHFAYEVXL7M7B2KOYCKJ5/action/replication_record"}},"created_at":"2026-07-05T11:16:06.420073+00:00","updated_at":"2026-07-05T11:16:06.420073+00:00"}