{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FQ4UUYJPSUZ7ITRX3UIYP37X2E","short_pith_number":"pith:FQ4UUYJP","schema_version":"1.0","canonical_sha256":"2c394a612f9533f44e37dd1187eff7d128b9782c8d65d34bb182de9b33e9aa4f","source":{"kind":"arxiv","id":"2407.07565","version":3},"attestation_state":"computed","paper":{"title":"On Leakage of Code Generation Evaluation Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandre Matton, Dennis Aumiller, Elena Tommasone, Ellen Gilsenan-McMahon, Jingyi He, Matthias Gall\\'e, Maxime Voisin, Milad Alizadeh, Raymond Ma, Tom Sherborne","submitted_at":"2024-07-10T11:50:20Z","abstract_excerpt":"In this paper, we consider contamination by code generation test sets, in particular in their use in modern large language models. We discuss three possible sources of such contamination and show findings supporting each of them: (i) direct data leakage, (ii) indirect data leakage through the use of synthetic data and (iii) overfitting to evaluation sets during model selection. To address this, we release Less Basic Python Problems (LBPP): an uncontaminated new benchmark of 161 prompts with their associated Python solutions. LBPP is released at https://huggingface.co/datasets/CohereForAI/lbpp "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07565","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-10T11:50:20Z","cross_cats_sorted":[],"title_canon_sha256":"271848190bd86b1796e026febfd492029f20353900adb7deb97ecf0e4339a1d9","abstract_canon_sha256":"c576e2c35a8ad04ffc405d27e36ab340aea7a4ee07c7a681d834abbcd04c999b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:13.125128Z","signature_b64":"vAGABJFiWbM+Kgm1qDysSzarFSXk/Z6yEJCdr1mDyr6qG5e7rqDCbuW/Bcb1KUjSBbin0UA4cY80nSEs1zXsDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c394a612f9533f44e37dd1187eff7d128b9782c8d65d34bb182de9b33e9aa4f","last_reissued_at":"2026-07-05T09:15:13.124627Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:13.124627Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Leakage of Code Generation Evaluation Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandre Matton, Dennis Aumiller, Elena Tommasone, Ellen Gilsenan-McMahon, Jingyi He, Matthias Gall\\'e, Maxime Voisin, Milad Alizadeh, Raymond Ma, Tom Sherborne","submitted_at":"2024-07-10T11:50:20Z","abstract_excerpt":"In this paper, we consider contamination by code generation test sets, in particular in their use in modern large language models. We discuss three possible sources of such contamination and show findings supporting each of them: (i) direct data leakage, (ii) indirect data leakage through the use of synthetic data and (iii) overfitting to evaluation sets during model selection. To address this, we release Less Basic Python Problems (LBPP): an uncontaminated new benchmark of 161 prompts with their associated Python solutions. LBPP is released at https://huggingface.co/datasets/CohereForAI/lbpp "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07565","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07565/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07565","created_at":"2026-07-05T09:15:13.124685+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07565v3","created_at":"2026-07-05T09:15:13.124685+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07565","created_at":"2026-07-05T09:15:13.124685+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQ4UUYJPSUZ7","created_at":"2026-07-05T09:15:13.124685+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQ4UUYJPSUZ7ITRX","created_at":"2026-07-05T09:15:13.124685+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQ4UUYJP","created_at":"2026-07-05T09:15:13.124685+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.05501","citing_title":"Towards Real-World Validity in Generative AI Benchmarks: Understanding and Designing Domain-Centered Evaluations for Journalism Practitioners","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E","json":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E.json","graph_json":"https://pith.science/api/pith-number/FQ4UUYJPSUZ7ITRX3UIYP37X2E/graph.json","events_json":"https://pith.science/api/pith-number/FQ4UUYJPSUZ7ITRX3UIYP37X2E/events.json","paper":"https://pith.science/paper/FQ4UUYJP"},"agent_actions":{"view_html":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E","download_json":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E.json","view_paper":"https://pith.science/paper/FQ4UUYJP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07565&json=true","fetch_graph":"https://pith.science/api/pith-number/FQ4UUYJPSUZ7ITRX3UIYP37X2E/graph.json","fetch_events":"https://pith.science/api/pith-number/FQ4UUYJPSUZ7ITRX3UIYP37X2E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E/action/storage_attestation","attest_author":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E/action/author_attestation","sign_citation":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E/action/citation_signature","submit_replication":"https://pith.science/pith/FQ4UUYJPSUZ7ITRX3UIYP37X2E/action/replication_record"}},"created_at":"2026-07-05T09:15:13.124685+00:00","updated_at":"2026-07-05T09:15:13.124685+00:00"}