{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WIDIQ5T5MPNKPXTWZYKLQSMSLL","short_pith_number":"pith:WIDIQ5T5","schema_version":"1.0","canonical_sha256":"b20688767d63daa7de76ce14b849925afb726822e8ea2cecc15c37b29a4b22f7","source":{"kind":"arxiv","id":"2411.03923","version":1},"attestation_state":"computed","paper":{"title":"Evaluation data contamination in LLMs: how do we measure it and (when) does it matter?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aaditya K. Singh, Andrew Poulton, David Esiobu, Dieuwke Hupkes, Gergely Szilvasy, Maria Lomeli, Muhammed Yusuf Kocyigit","submitted_at":"2024-11-06T13:54:08Z","abstract_excerpt":"Hampering the interpretation of benchmark scores, evaluation data contamination has become a growing concern in the evaluation of LLMs, and an active area of research studies its effects. While evaluation data contamination is easily understood intuitively, it is surprisingly difficult to define precisely which samples should be considered contaminated and, consequently, how it impacts benchmark scores. We propose that these questions should be addressed together and that contamination metrics can be assessed based on whether models benefit from the examples they mark contaminated. We propose "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.03923","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-06T13:54:08Z","cross_cats_sorted":[],"title_canon_sha256":"ddd1a9fbc02ff2f16b988a434a178242b4450ff825a653c19e23ab4b542861bf","abstract_canon_sha256":"bffd0af25f0ca46d84bed88893932c9704faf9d1c49e1432fac4db5368f2e354"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:58.227009Z","signature_b64":"+eG4ME4NKH5PXvZyREShQPcs3EUrdK1a1ieyG+/MwjePHdwonl1tMqKrYUYBjgmAeBTaI75I8ANXs0s4iKl+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b20688767d63daa7de76ce14b849925afb726822e8ea2cecc15c37b29a4b22f7","last_reissued_at":"2026-07-05T09:31:58.226521Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:58.226521Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation data contamination in LLMs: how do we measure it and (when) does it matter?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aaditya K. Singh, Andrew Poulton, David Esiobu, Dieuwke Hupkes, Gergely Szilvasy, Maria Lomeli, Muhammed Yusuf Kocyigit","submitted_at":"2024-11-06T13:54:08Z","abstract_excerpt":"Hampering the interpretation of benchmark scores, evaluation data contamination has become a growing concern in the evaluation of LLMs, and an active area of research studies its effects. While evaluation data contamination is easily understood intuitively, it is surprisingly difficult to define precisely which samples should be considered contaminated and, consequently, how it impacts benchmark scores. We propose that these questions should be addressed together and that contamination metrics can be assessed based on whether models benefit from the examples they mark contaminated. We propose "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03923","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.03923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.03923","created_at":"2026-07-05T09:31:58.226582+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.03923v1","created_at":"2026-07-05T09:31:58.226582+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03923","created_at":"2026-07-05T09:31:58.226582+00:00"},{"alias_kind":"pith_short_12","alias_value":"WIDIQ5T5MPNK","created_at":"2026-07-05T09:31:58.226582+00:00"},{"alias_kind":"pith_short_16","alias_value":"WIDIQ5T5MPNKPXTW","created_at":"2026-07-05T09:31:58.226582+00:00"},{"alias_kind":"pith_short_8","alias_value":"WIDIQ5T5","created_at":"2026-07-05T09:31:58.226582+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00890","citing_title":"MultiSynt/MT: Trillion-Token Multi-Parallel Pre-Training Data Translated Across 36 Languages","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31991","citing_title":"Amplifying Membership Signal Through Chained Regeneration","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24213","citing_title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19590","citing_title":"Position: AI Evaluations Should be Grounded on a Theory of Capability","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2406.19314","citing_title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17842","citing_title":"QuickScope: Certifying Hard Questions in Dynamic LLM Benchmarks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06865","citing_title":"Dataset Watermarking for Closed LLMs with Provable Detection","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02442","citing_title":"Measuring AI Reasoning: A Guide for Researchers","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL","json":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL.json","graph_json":"https://pith.science/api/pith-number/WIDIQ5T5MPNKPXTWZYKLQSMSLL/graph.json","events_json":"https://pith.science/api/pith-number/WIDIQ5T5MPNKPXTWZYKLQSMSLL/events.json","paper":"https://pith.science/paper/WIDIQ5T5"},"agent_actions":{"view_html":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL","download_json":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL.json","view_paper":"https://pith.science/paper/WIDIQ5T5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.03923&json=true","fetch_graph":"https://pith.science/api/pith-number/WIDIQ5T5MPNKPXTWZYKLQSMSLL/graph.json","fetch_events":"https://pith.science/api/pith-number/WIDIQ5T5MPNKPXTWZYKLQSMSLL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL/action/storage_attestation","attest_author":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL/action/author_attestation","sign_citation":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL/action/citation_signature","submit_replication":"https://pith.science/pith/WIDIQ5T5MPNKPXTWZYKLQSMSLL/action/replication_record"}},"created_at":"2026-07-05T09:31:58.226582+00:00","updated_at":"2026-07-05T09:31:58.226582+00:00"}