{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LN3BKEMSW77YZK5JCED47DASGJ","short_pith_number":"pith:LN3BKEMS","schema_version":"1.0","canonical_sha256":"5b76151192b7ff8caba91107cf8c12327c92720b5cb79a046f52cf690fb574ff","source":{"kind":"arxiv","id":"2410.18966","version":3},"attestation_state":"computed","paper":{"title":"Does Data Contamination Detection Work (Well) for LLMs? A Survey and Evaluation on Detection Assumptions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fei Xia, Meliha Yetisgen, Ozlem Uzuner, Yujuan Fu","submitted_at":"2024-10-24T17:58:22Z","abstract_excerpt":"Large language models (LLMs) have demonstrated great performance across various benchmarks, showing potential as general-purpose task solvers. However, as LLMs are typically trained on vast amounts of data, a significant concern in their evaluation is data contamination, where overlap between training data and evaluation datasets inflates performance assessments. Multiple approaches have been developed to identify data contamination. These approaches rely on specific assumptions that may not hold universally across different settings. To bridge this gap, we systematically review 50 papers on d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18966","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T17:58:22Z","cross_cats_sorted":[],"title_canon_sha256":"a2d857cc9a6eafc279cb95c77516c99af281aba569a33eb7ff9a5644d76d5357","abstract_canon_sha256":"f266f07177ad41eae390bb7eb2a6ecffb64e0cf086b5c4b4a3502a9939033966"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:02.142294Z","signature_b64":"+qv52E5yiJaAcHNAsR2W17eydcWmPeejZknltTemre4K1ciYATw+iYPw8hT3IPxh/iDkDQb6PYGVktb0dvNkBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b76151192b7ff8caba91107cf8c12327c92720b5cb79a046f52cf690fb574ff","last_reissued_at":"2026-07-05T11:01:02.141795Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:02.141795Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Data Contamination Detection Work (Well) for LLMs? A Survey and Evaluation on Detection Assumptions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fei Xia, Meliha Yetisgen, Ozlem Uzuner, Yujuan Fu","submitted_at":"2024-10-24T17:58:22Z","abstract_excerpt":"Large language models (LLMs) have demonstrated great performance across various benchmarks, showing potential as general-purpose task solvers. However, as LLMs are typically trained on vast amounts of data, a significant concern in their evaluation is data contamination, where overlap between training data and evaluation datasets inflates performance assessments. Multiple approaches have been developed to identify data contamination. These approaches rely on specific assumptions that may not hold universally across different settings. To bridge this gap, we systematically review 50 papers on d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18966","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18966/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18966","created_at":"2026-07-05T11:01:02.141851+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18966v3","created_at":"2026-07-05T11:01:02.141851+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18966","created_at":"2026-07-05T11:01:02.141851+00:00"},{"alias_kind":"pith_short_12","alias_value":"LN3BKEMSW77Y","created_at":"2026-07-05T11:01:02.141851+00:00"},{"alias_kind":"pith_short_16","alias_value":"LN3BKEMSW77YZK5J","created_at":"2026-07-05T11:01:02.141851+00:00"},{"alias_kind":"pith_short_8","alias_value":"LN3BKEMS","created_at":"2026-07-05T11:01:02.141851+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ","json":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ.json","graph_json":"https://pith.science/api/pith-number/LN3BKEMSW77YZK5JCED47DASGJ/graph.json","events_json":"https://pith.science/api/pith-number/LN3BKEMSW77YZK5JCED47DASGJ/events.json","paper":"https://pith.science/paper/LN3BKEMS"},"agent_actions":{"view_html":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ","download_json":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ.json","view_paper":"https://pith.science/paper/LN3BKEMS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18966&json=true","fetch_graph":"https://pith.science/api/pith-number/LN3BKEMSW77YZK5JCED47DASGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/LN3BKEMSW77YZK5JCED47DASGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ/action/storage_attestation","attest_author":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ/action/author_attestation","sign_citation":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ/action/citation_signature","submit_replication":"https://pith.science/pith/LN3BKEMSW77YZK5JCED47DASGJ/action/replication_record"}},"created_at":"2026-07-05T11:01:02.141851+00:00","updated_at":"2026-07-05T11:01:02.141851+00:00"}