{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PNHPTCMOGDQ7VFOU4SMAABS6EA","short_pith_number":"pith:PNHPTCMO","schema_version":"1.0","canonical_sha256":"7b4ef9898e30e1fa95d4e49800065e20240cbc0e444b5982b90d7a67ce963910","source":{"kind":"arxiv","id":"2311.01964","version":1},"attestation_state":"computed","paper":{"title":"Don't Make Your LLM an Evaluation Benchmark Cheater","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiawei Han, Ji-Rong Wen, Kun Zhou, Wayne Xin Zhao, Wentong Chen, Xu Chen, Yankai Lin, Yutao Zhu, Zhipeng Chen","submitted_at":"2023-11-03T14:59:54Z","abstract_excerpt":"Large language models~(LLMs) have greatly advanced the frontiers of artificial intelligence, attaining remarkable improvement in model capacity. To assess the model performance, a typical approach is to construct evaluation benchmarks for measuring the ability level of LLMs in different aspects. Despite that a number of high-quality benchmarks have been released, the concerns about the appropriate use of these benchmarks and the fair comparison of different models are increasingly growing. Considering these concerns, in this paper, we discuss the potential risk and impact of inappropriately us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.01964","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-03T14:59:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b51e749f13a6ed8b8dc147dbe8f97d8d27ff1e9506bfd9db2f7708fb0985174e","abstract_canon_sha256":"ee518a6f37115a8465966dbf5af2dcae55d472bacbb6ec64dfa54f7f377ab68a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:48.050903Z","signature_b64":"pBcZ9+serCtHozq21dDOj6MxZYYnP35m5XTKv8MjbU4pZ5Anw2b/s6yuRRCc4kidV1p2K0zTXtIegXnU8ri+BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b4ef9898e30e1fa95d4e49800065e20240cbc0e444b5982b90d7a67ce963910","last_reissued_at":"2026-07-05T07:08:48.050422Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:48.050422Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't Make Your LLM an Evaluation Benchmark Cheater","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiawei Han, Ji-Rong Wen, Kun Zhou, Wayne Xin Zhao, Wentong Chen, Xu Chen, Yankai Lin, Yutao Zhu, Zhipeng Chen","submitted_at":"2023-11-03T14:59:54Z","abstract_excerpt":"Large language models~(LLMs) have greatly advanced the frontiers of artificial intelligence, attaining remarkable improvement in model capacity. To assess the model performance, a typical approach is to construct evaluation benchmarks for measuring the ability level of LLMs in different aspects. Despite that a number of high-quality benchmarks have been released, the concerns about the appropriate use of these benchmarks and the fair comparison of different models are increasingly growing. Considering these concerns, in this paper, we discuss the potential risk and impact of inappropriately us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.01964","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.01964/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.01964","created_at":"2026-07-05T07:08:48.050482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.01964v1","created_at":"2026-07-05T07:08:48.050482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.01964","created_at":"2026-07-05T07:08:48.050482+00:00"},{"alias_kind":"pith_short_12","alias_value":"PNHPTCMOGDQ7","created_at":"2026-07-05T07:08:48.050482+00:00"},{"alias_kind":"pith_short_16","alias_value":"PNHPTCMOGDQ7VFOU","created_at":"2026-07-05T07:08:48.050482+00:00"},{"alias_kind":"pith_short_8","alias_value":"PNHPTCMO","created_at":"2026-07-05T07:08:48.050482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08009","citing_title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","ref_index":208,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11166","citing_title":"Flaws in the LLM Automation Narrative","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28863","citing_title":"Defeat Devices in AI Systems","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30916","citing_title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03952","citing_title":"Bridging Language and Items for Retrieval and Recommendation: Benchmarking LLMs as Semantic Encoders","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04244","citing_title":"Benchmark Data Contamination of Large Language Models: A Survey","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21543","citing_title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","ref_index":181,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20909","citing_title":"LogitTrace: Detecting Benchmark Contamination via Layerwise Logit Trajectories","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2406.16860","citing_title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","ref_index":155,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12910","citing_title":"SciCoQA: Quality Assurance for Scientific Paper--Code Alignment","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18916","citing_title":"Agentic Business Process Management: A Research Manifesto","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08838","citing_title":"Generating Leakage-Free Benchmarks for Robust RAG Evaluation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24819","citing_title":"Programming with Data: Test-Driven Data Engineering for Self-Improving LLMs from Raw Corpora","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24544","citing_title":"STELLAR-E: a Synthetic, Tailored, End-to-end LLM Application Rigorous Evaluator","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23067","citing_title":"Training a General Purpose Automated Red Teaming Model","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20273","citing_title":"ActuBench: A Multi-Agent LLM Pipeline for Generation and Evaluation of Actuarial Reasoning Tasks","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07650","citing_title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06802","citing_title":"Riemann-Bench: A Benchmark for Moonshot Mathematics","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04815","citing_title":"LiveFact: A Dynamic, Time-Aware Benchmark for LLM-Driven Fake News Detection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":223,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA","json":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA.json","graph_json":"https://pith.science/api/pith-number/PNHPTCMOGDQ7VFOU4SMAABS6EA/graph.json","events_json":"https://pith.science/api/pith-number/PNHPTCMOGDQ7VFOU4SMAABS6EA/events.json","paper":"https://pith.science/paper/PNHPTCMO"},"agent_actions":{"view_html":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA","download_json":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA.json","view_paper":"https://pith.science/paper/PNHPTCMO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.01964&json=true","fetch_graph":"https://pith.science/api/pith-number/PNHPTCMOGDQ7VFOU4SMAABS6EA/graph.json","fetch_events":"https://pith.science/api/pith-number/PNHPTCMOGDQ7VFOU4SMAABS6EA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA/action/storage_attestation","attest_author":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA/action/author_attestation","sign_citation":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA/action/citation_signature","submit_replication":"https://pith.science/pith/PNHPTCMOGDQ7VFOU4SMAABS6EA/action/replication_record"}},"created_at":"2026-07-05T07:08:48.050482+00:00","updated_at":"2026-07-05T07:08:48.050482+00:00"}