{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WE255WASGWL6QVPJIW63VQX4AV","short_pith_number":"pith:WE255WAS","schema_version":"1.0","canonical_sha256":"b135ded8123597e855e945bdbac2fc05480f55076a356d570a3bcc289c14e765","source":{"kind":"arxiv","id":"2412.03597","version":1},"attestation_state":"computed","paper":{"title":"The Vulnerability of Language Model Benchmarks: Do They Accurately Reflect True LLM Performance?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Ayushi Agarwal, Eishkaran Singh, Sourav Banerjee","submitted_at":"2024-12-02T20:49:21Z","abstract_excerpt":"The pursuit of leaderboard rankings in Large Language Models (LLMs) has created a fundamental paradox: models excel at standardized tests while failing to demonstrate genuine language understanding and adaptability. Our systematic analysis of NLP evaluation frameworks reveals pervasive vulnerabilities across the evaluation spectrum, from basic metrics to complex benchmarks like GLUE and MMLU. These vulnerabilities manifest through benchmark exploitation, dataset contamination, and evaluation bias, creating a false perception of progress in language understanding capabilities. Through extensive"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.03597","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-02T20:49:21Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"32513eeed06c060390a6bdb2aa896248500b9f531472cf2063a3802db2b716b9","abstract_canon_sha256":"969e24221cd988f4aba4ee5728ce5135c22cb96079684bf2e828e62f600dd027"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:38.487999Z","signature_b64":"a00T+IaMPjf02+vS14UvQqjhYR9ZUvQeOGrNqvipmk2fyOT90SVaAS44TFVF2xNtCAco+2ZwqPU/teB30kYNDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b135ded8123597e855e945bdbac2fc05480f55076a356d570a3bcc289c14e765","last_reissued_at":"2026-07-05T09:44:38.487520Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:38.487520Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Vulnerability of Language Model Benchmarks: Do They Accurately Reflect True LLM Performance?","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Ayushi Agarwal, Eishkaran Singh, Sourav Banerjee","submitted_at":"2024-12-02T20:49:21Z","abstract_excerpt":"The pursuit of leaderboard rankings in Large Language Models (LLMs) has created a fundamental paradox: models excel at standardized tests while failing to demonstrate genuine language understanding and adaptability. Our systematic analysis of NLP evaluation frameworks reveals pervasive vulnerabilities across the evaluation spectrum, from basic metrics to complex benchmarks like GLUE and MMLU. These vulnerabilities manifest through benchmark exploitation, dataset contamination, and evaluation bias, creating a false perception of progress in language understanding capabilities. Through extensive"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.03597","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.03597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.03597","created_at":"2026-07-05T09:44:38.487578+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.03597v1","created_at":"2026-07-05T09:44:38.487578+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.03597","created_at":"2026-07-05T09:44:38.487578+00:00"},{"alias_kind":"pith_short_12","alias_value":"WE255WASGWL6","created_at":"2026-07-05T09:44:38.487578+00:00"},{"alias_kind":"pith_short_16","alias_value":"WE255WASGWL6QVPJ","created_at":"2026-07-05T09:44:38.487578+00:00"},{"alias_kind":"pith_short_8","alias_value":"WE255WAS","created_at":"2026-07-05T09:44:38.487578+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30018","citing_title":"Latent Performance Profiling of Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05452","citing_title":"LLMEval-Fair: A Large-Scale Longitudinal Study on Robust and Fair Evaluation of Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23023","citing_title":"Deceive, Detect, and Disclose: Large Language Models Play Mini-Mafia","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06989","citing_title":"Human-aligned AI Model Cards with Weighted Hierarchy Architecture","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23067","citing_title":"Training a General Purpose Automated Red Teaming Model","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13371","citing_title":"Empirical Evidence of Complexity-Induced Limits in Large Language Models on Finite Discrete State-Space Problems with Explicit Validity Constraints","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV","json":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV.json","graph_json":"https://pith.science/api/pith-number/WE255WASGWL6QVPJIW63VQX4AV/graph.json","events_json":"https://pith.science/api/pith-number/WE255WASGWL6QVPJIW63VQX4AV/events.json","paper":"https://pith.science/paper/WE255WAS"},"agent_actions":{"view_html":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV","download_json":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV.json","view_paper":"https://pith.science/paper/WE255WAS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.03597&json=true","fetch_graph":"https://pith.science/api/pith-number/WE255WASGWL6QVPJIW63VQX4AV/graph.json","fetch_events":"https://pith.science/api/pith-number/WE255WASGWL6QVPJIW63VQX4AV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV/action/storage_attestation","attest_author":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV/action/author_attestation","sign_citation":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV/action/citation_signature","submit_replication":"https://pith.science/pith/WE255WASGWL6QVPJIW63VQX4AV/action/replication_record"}},"created_at":"2026-07-05T09:44:38.487578+00:00","updated_at":"2026-07-05T09:44:38.487578+00:00"}