{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MA5GLWWYKZVARUTGX4HI6CYFFI","short_pith_number":"pith:MA5GLWWY","schema_version":"1.0","canonical_sha256":"603a65dad8566a08d266bf0e8f0b052a2c0177cd346721c5fa5c7876b9521af8","source":{"kind":"arxiv","id":"2508.15361","version":1},"attestation_state":"computed","paper":{"title":"A Survey on Large Language Model Benchmarks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingli Wang, Chengming Li, Guhong Chen, Le Sun, Liyang Fan, Min Yang, Qiyao Wang, Ruifeng Xu, Shiwen Ni, Shuaimin Li, Siyi Li, Xingjian Wang, Xuanang Chen, Yifan Zhang","submitted_at":"2025-08-21T08:43:35Z","abstract_excerpt":"In recent years, with the rapid development of the depth and breadth of large language models' capabilities, various corresponding evaluation benchmarks have been emerging in increasing numbers. As a quantitative assessment tool for model performance, benchmarks are not only a core means to measure model capabilities but also a key element in guiding the direction of model development and promoting technological innovation. We systematically review the current status and development of large language model benchmarks for the first time, categorizing 283 representative benchmarks into three cat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.15361","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-21T08:43:35Z","cross_cats_sorted":[],"title_canon_sha256":"9e93627e07a757f1982b253835f51e472b59c515fb3a76e72838fde2fa11f273","abstract_canon_sha256":"3d80125a5b06a98bdc19c363ae99313271b353b4485a5aa9c3ecf4d7eafff9b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:57:10.913388Z","signature_b64":"w6VaxZvC34ImcWgF/VjHZ7qzWZ9DYYhOnXjKrm90+YWvcGt0RlCb+NS3oDtinBolQaO90DwbvdXajpN/bTI9Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"603a65dad8566a08d266bf0e8f0b052a2c0177cd346721c5fa5c7876b9521af8","last_reissued_at":"2026-07-05T11:57:10.912908Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:57:10.912908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Large Language Model Benchmarks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bingli Wang, Chengming Li, Guhong Chen, Le Sun, Liyang Fan, Min Yang, Qiyao Wang, Ruifeng Xu, Shiwen Ni, Shuaimin Li, Siyi Li, Xingjian Wang, Xuanang Chen, Yifan Zhang","submitted_at":"2025-08-21T08:43:35Z","abstract_excerpt":"In recent years, with the rapid development of the depth and breadth of large language models' capabilities, various corresponding evaluation benchmarks have been emerging in increasing numbers. As a quantitative assessment tool for model performance, benchmarks are not only a core means to measure model capabilities but also a key element in guiding the direction of model development and promoting technological innovation. We systematically review the current status and development of large language model benchmarks for the first time, categorizing 283 representative benchmarks into three cat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.15361","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.15361/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.15361","created_at":"2026-07-05T11:57:10.912964+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.15361v1","created_at":"2026-07-05T11:57:10.912964+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.15361","created_at":"2026-07-05T11:57:10.912964+00:00"},{"alias_kind":"pith_short_12","alias_value":"MA5GLWWYKZVA","created_at":"2026-07-05T11:57:10.912964+00:00"},{"alias_kind":"pith_short_16","alias_value":"MA5GLWWYKZVARUTG","created_at":"2026-07-05T11:57:10.912964+00:00"},{"alias_kind":"pith_short_8","alias_value":"MA5GLWWY","created_at":"2026-07-05T11:57:10.912964+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09809","citing_title":"Evaluation Cards: An Interpretive Layer for AI Evaluation Reporting","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00664","citing_title":"YOMI-Bench: A Benchmark for Evaluating Kanji Reading and Phonological Understanding of LLMs for Japanese","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30459","citing_title":"Can LLM Teams Play What? Where? When?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30738","citing_title":"MAVEN: Improving Generalization in Agentic Tool Calling","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23628","citing_title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2602.03433","citing_title":"When control meets large language models: From words to dynamics","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15208","citing_title":"Quantization Undoes Alignment: Bias Emergence in Compressed LLMs Across Models and Precision Levels","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15393","citing_title":"LPDS: Evaluating LLM Robustness Through Logic-Preserving Difficulty Scaling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02188","citing_title":"Reasoning in a Combinatorial and Constrained World: Benchmarking LLMs on Natural-Language Combinatorial Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09588","citing_title":"Efficient Ensemble Selection from Binary and Pairwise Feedback","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12473","citing_title":"Designing for Error Recovery in Human-Robot Interaction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13371","citing_title":"Empirical Evidence of Complexity-Induced Limits in Large Language Models on Finite Discrete State-Space Problems with Explicit Validity Constraints","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI","json":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI.json","graph_json":"https://pith.science/api/pith-number/MA5GLWWYKZVARUTGX4HI6CYFFI/graph.json","events_json":"https://pith.science/api/pith-number/MA5GLWWYKZVARUTGX4HI6CYFFI/events.json","paper":"https://pith.science/paper/MA5GLWWY"},"agent_actions":{"view_html":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI","download_json":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI.json","view_paper":"https://pith.science/paper/MA5GLWWY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.15361&json=true","fetch_graph":"https://pith.science/api/pith-number/MA5GLWWYKZVARUTGX4HI6CYFFI/graph.json","fetch_events":"https://pith.science/api/pith-number/MA5GLWWYKZVARUTGX4HI6CYFFI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI/action/storage_attestation","attest_author":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI/action/author_attestation","sign_citation":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI/action/citation_signature","submit_replication":"https://pith.science/pith/MA5GLWWYKZVARUTGX4HI6CYFFI/action/replication_record"}},"created_at":"2026-07-05T11:57:10.912964+00:00","updated_at":"2026-07-05T11:57:10.912964+00:00"}