{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KCWBYPZBYJUGCOSTVDPB4JL5FY","short_pith_number":"pith:KCWBYPZB","schema_version":"1.0","canonical_sha256":"50ac1c3f21c268613a53a8de1e257d2e37cefff2950309b4cf9738514b325c73","source":{"kind":"arxiv","id":"2311.04850","version":2},"attestation_state":"computed","paper":{"title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ion Stoica, Joseph E. Gonzalez, Lianmin Zheng, Shuo Yang, Wei-Lin Chiang","submitted_at":"2023-11-08T17:35:20Z","abstract_excerpt":"Large language models are increasingly trained on all the data ever produced by humans. Many have raised concerns about the trustworthiness of public benchmarks due to potential contamination in pre-training or fine-tuning datasets. While most data decontamination efforts apply string matching (e.g., n-gram overlap) to remove benchmark data, we show that these methods are insufficient, and simple variations of test data (e.g., paraphrasing, translation) can easily bypass these decontamination measures. Furthermore, we demonstrate that if such variation of test data is not eliminated, a 13B mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.04850","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-08T17:35:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8869a3877ed28ad6e93e8c1b70dcae80af2c9f172bf3621186c66bb808d6a83f","abstract_canon_sha256":"f3f5f1af840fb19c3bb7f519b42cddd3c8a6f37053e3da5554e9068a6f884987"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:37.536644Z","signature_b64":"7+67O0SCti73+QAWW3OczlCgDvVhQxnaQkkFkvGdxx0TCcNDibI/w2uqZaqrrtA9N11oq9LqvWlFTKIoTTm5Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50ac1c3f21c268613a53a8de1e257d2e37cefff2950309b4cf9738514b325c73","last_reissued_at":"2026-07-05T07:11:37.536112Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:37.536112Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ion Stoica, Joseph E. Gonzalez, Lianmin Zheng, Shuo Yang, Wei-Lin Chiang","submitted_at":"2023-11-08T17:35:20Z","abstract_excerpt":"Large language models are increasingly trained on all the data ever produced by humans. Many have raised concerns about the trustworthiness of public benchmarks due to potential contamination in pre-training or fine-tuning datasets. While most data decontamination efforts apply string matching (e.g., n-gram overlap) to remove benchmark data, we show that these methods are insufficient, and simple variations of test data (e.g., paraphrasing, translation) can easily bypass these decontamination measures. Furthermore, we demonstrate that if such variation of test data is not eliminated, a 13B mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.04850","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.04850/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.04850","created_at":"2026-07-05T07:11:37.536171+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.04850v2","created_at":"2026-07-05T07:11:37.536171+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.04850","created_at":"2026-07-05T07:11:37.536171+00:00"},{"alias_kind":"pith_short_12","alias_value":"KCWBYPZBYJUG","created_at":"2026-07-05T07:11:37.536171+00:00"},{"alias_kind":"pith_short_16","alias_value":"KCWBYPZBYJUGCOST","created_at":"2026-07-05T07:11:37.536171+00:00"},{"alias_kind":"pith_short_8","alias_value":"KCWBYPZB","created_at":"2026-07-05T07:11:37.536171+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11909","citing_title":"Embodied-BenchClaw: An Autonomous Multi-Agent System for Embodied Spatial Intelligence Benchmark Construction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12385","citing_title":"Which Models Are Our Models Built On? Auditing Invisible Dependencies in Modern LLMs","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07053","citing_title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24079","citing_title":"TRACER: A Semantic-Aware Framework for Fine-Grained Contamination Detection in Code LLMs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24213","citing_title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24661","citing_title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26161","citing_title":"TSFMAudit: Data Contamination Auditing in Forecasting Time Series Foundation Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29815","citing_title":"SrDetection: A Self-Referential Framework for Data Leakage Detection in Code Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04244","citing_title":"Benchmark Data Contamination of Large Language Models: A Survey","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21856","citing_title":"The Illusion of Reasoning: Exposing Evasive Data Contamination in LLMs via Zero-CoT Truncation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21543","citing_title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","ref_index":172,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19999","citing_title":"LLM Benchmark Datasets Should Be Contamination-Resistant","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22359","citing_title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20909","citing_title":"LogitTrace: Detecting Benchmark Contamination via Layerwise Logit Trajectories","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":208,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12673","citing_title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11501","citing_title":"Decaf: Improving Neural Decompilation with Automatic Feedback and Search","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10448","citing_title":"Can Agent Benchmarks Support Their Scores? Evidence-Supported Bounds for Interactive-Agent Evaluation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24712","citing_title":"When Prompt Under-Specification Improves Code Correctness: An Exploratory Study of Prompt Wording and Structure Effects on LLM-Based Code Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04312","citing_title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09251","citing_title":"DRBENCHER: Can Your Agent Identify the Entity, Retrieve Its Properties and Do the Math?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2406.12793","citing_title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY","json":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY.json","graph_json":"https://pith.science/api/pith-number/KCWBYPZBYJUGCOSTVDPB4JL5FY/graph.json","events_json":"https://pith.science/api/pith-number/KCWBYPZBYJUGCOSTVDPB4JL5FY/events.json","paper":"https://pith.science/paper/KCWBYPZB"},"agent_actions":{"view_html":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY","download_json":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY.json","view_paper":"https://pith.science/paper/KCWBYPZB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.04850&json=true","fetch_graph":"https://pith.science/api/pith-number/KCWBYPZBYJUGCOSTVDPB4JL5FY/graph.json","fetch_events":"https://pith.science/api/pith-number/KCWBYPZBYJUGCOSTVDPB4JL5FY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY/action/storage_attestation","attest_author":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY/action/author_attestation","sign_citation":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY/action/citation_signature","submit_replication":"https://pith.science/pith/KCWBYPZBYJUGCOSTVDPB4JL5FY/action/replication_record"}},"created_at":"2026-07-05T07:11:37.536171+00:00","updated_at":"2026-07-05T07:11:37.536171+00:00"}