{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GPJSTAHIITKQH5F4J6UYH5RZFY","short_pith_number":"pith:GPJSTAHI","schema_version":"1.0","canonical_sha256":"33d32980e844d503f4bc4fa983f6392e3f633b22c52b1d72a6400bbfe3dbf56d","source":{"kind":"arxiv","id":"2410.03249","version":4},"attestation_state":"computed","paper":{"title":"How Much Can We Forget about Data Contamination?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Sebastian Bordt, Suraj Srinivas, Ulrike von Luxburg, Valentyn Boreiko","submitted_at":"2024-10-04T09:14:11Z","abstract_excerpt":"The leakage of benchmark data into the training data has emerged as a significant challenge for evaluating the capabilities of large language models (LLMs). In this work, we challenge the common assumption that small-scale contamination renders benchmark evaluations invalid. First, we experimentally quantify the magnitude of benchmark overfitting based on scaling along three dimensions: The number of model parameters (up to 1.6B), the number of times an example is seen (up to 144), and the number of training tokens (up to 40B). If model and data follow the Chinchilla scaling laws, minor contam"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03249","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-04T09:14:11Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cdc9bb92f406c159a2b41c138081a5fa1a8ec632cda8519b2befe15db0285926","abstract_canon_sha256":"c5cc7f4af2600a69ef0837b45e30af48bca5d9981ee14739039dc17d3eb7fc66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:36.403556Z","signature_b64":"piu86o6j6AhUf7PVN9chPX/1gKyga3l9rR+ls84gebNfb+6kE3qmzoDaAv3+4qmwYBCxCZTMuSGtZo5hvkgOBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"33d32980e844d503f4bc4fa983f6392e3f633b22c52b1d72a6400bbfe3dbf56d","last_reissued_at":"2026-07-05T11:21:36.403012Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:36.403012Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Much Can We Forget about Data Contamination?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Sebastian Bordt, Suraj Srinivas, Ulrike von Luxburg, Valentyn Boreiko","submitted_at":"2024-10-04T09:14:11Z","abstract_excerpt":"The leakage of benchmark data into the training data has emerged as a significant challenge for evaluating the capabilities of large language models (LLMs). In this work, we challenge the common assumption that small-scale contamination renders benchmark evaluations invalid. First, we experimentally quantify the magnitude of benchmark overfitting based on scaling along three dimensions: The number of model parameters (up to 1.6B), the number of times an example is seen (up to 144), and the number of training tokens (up to 40B). If model and data follow the Chinchilla scaling laws, minor contam"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03249","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03249/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03249","created_at":"2026-07-05T11:21:36.403071+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03249v4","created_at":"2026-07-05T11:21:36.403071+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03249","created_at":"2026-07-05T11:21:36.403071+00:00"},{"alias_kind":"pith_short_12","alias_value":"GPJSTAHIITKQ","created_at":"2026-07-05T11:21:36.403071+00:00"},{"alias_kind":"pith_short_16","alias_value":"GPJSTAHIITKQH5F4","created_at":"2026-07-05T11:21:36.403071+00:00"},{"alias_kind":"pith_short_8","alias_value":"GPJSTAHI","created_at":"2026-07-05T11:21:36.403071+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08465","citing_title":"From Safety Risk to Design Principle: Peer-Preservation in Multi-Agent LLM Systems and Its Implications for Orchestrated Democratic Discourse Analysis","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY","json":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY.json","graph_json":"https://pith.science/api/pith-number/GPJSTAHIITKQH5F4J6UYH5RZFY/graph.json","events_json":"https://pith.science/api/pith-number/GPJSTAHIITKQH5F4J6UYH5RZFY/events.json","paper":"https://pith.science/paper/GPJSTAHI"},"agent_actions":{"view_html":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY","download_json":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY.json","view_paper":"https://pith.science/paper/GPJSTAHI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03249&json=true","fetch_graph":"https://pith.science/api/pith-number/GPJSTAHIITKQH5F4J6UYH5RZFY/graph.json","fetch_events":"https://pith.science/api/pith-number/GPJSTAHIITKQH5F4J6UYH5RZFY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY/action/storage_attestation","attest_author":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY/action/author_attestation","sign_citation":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY/action/citation_signature","submit_replication":"https://pith.science/pith/GPJSTAHIITKQH5F4J6UYH5RZFY/action/replication_record"}},"created_at":"2026-07-05T11:21:36.403071+00:00","updated_at":"2026-07-05T11:21:36.403071+00:00"}