{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LLAVIIL3XP4PDQA7KK7EP4CYRX","short_pith_number":"pith:LLAVIIL3","schema_version":"1.0","canonical_sha256":"5ac154217bbbf8f1c01f52be47f0588dd51139bcc85c68d87fb16361c08b9ccf","source":{"kind":"arxiv","id":"2502.14318","version":1},"attestation_state":"computed","paper":{"title":"Line Goes Up? Inherent Limitations of Benchmarks for Evaluating Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"James Fodor","submitted_at":"2025-02-20T07:13:29Z","abstract_excerpt":"Large language models (LLMs) regularly demonstrate new and impressive performance on a wide range of language, knowledge, and reasoning benchmarks. Such rapid progress has led many commentators to argue that LLM general cognitive capabilities have likewise rapidly improved, with the implication that such models are becoming progressively more capable on various real-world tasks. Here I summarise theoretical and empirical considerations to challenge this narrative. I argue that inherent limitations with the benchmarking paradigm, along with specific limitations of existing benchmarks, render be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14318","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-20T07:13:29Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"fdc1708e6db09d40bf10e8c3751088b182bd1bdc41de5b84b03fe710b61a2510","abstract_canon_sha256":"137790d71442ca9692f606205488652b089351427756396bde285fc76303f185"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:24.302676Z","signature_b64":"P3eJPFkES1+LgfJzEccMKhRiVE/h/HzM35uBqCwy9YLYF0HzVajBD10Se+gh2pG2xmec1wRsWfH/B4dDoQIsCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ac154217bbbf8f1c01f52be47f0588dd51139bcc85c68d87fb16361c08b9ccf","last_reissued_at":"2026-07-05T10:17:24.302106Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:24.302106Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Line Goes Up? Inherent Limitations of Benchmarks for Evaluating Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"James Fodor","submitted_at":"2025-02-20T07:13:29Z","abstract_excerpt":"Large language models (LLMs) regularly demonstrate new and impressive performance on a wide range of language, knowledge, and reasoning benchmarks. Such rapid progress has led many commentators to argue that LLM general cognitive capabilities have likewise rapidly improved, with the implication that such models are becoming progressively more capable on various real-world tasks. Here I summarise theoretical and empirical considerations to challenge this narrative. I argue that inherent limitations with the benchmarking paradigm, along with specific limitations of existing benchmarks, render be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14318","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14318/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14318","created_at":"2026-07-05T10:17:24.302175+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14318v1","created_at":"2026-07-05T10:17:24.302175+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14318","created_at":"2026-07-05T10:17:24.302175+00:00"},{"alias_kind":"pith_short_12","alias_value":"LLAVIIL3XP4P","created_at":"2026-07-05T10:17:24.302175+00:00"},{"alias_kind":"pith_short_16","alias_value":"LLAVIIL3XP4PDQA7","created_at":"2026-07-05T10:17:24.302175+00:00"},{"alias_kind":"pith_short_8","alias_value":"LLAVIIL3","created_at":"2026-07-05T10:17:24.302175+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00288","citing_title":"Model-Native Computing Architecture: Envisioning Future System Architecture Through the Lens of Computer Architecture","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30018","citing_title":"Latent Performance Profiling of Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14164","citing_title":"Unsteady Metrics and Benchmarking Cultures of AI Model Builders","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08942","citing_title":"Decomposing and Steering Functional Metacognition in Large Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX","json":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX.json","graph_json":"https://pith.science/api/pith-number/LLAVIIL3XP4PDQA7KK7EP4CYRX/graph.json","events_json":"https://pith.science/api/pith-number/LLAVIIL3XP4PDQA7KK7EP4CYRX/events.json","paper":"https://pith.science/paper/LLAVIIL3"},"agent_actions":{"view_html":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX","download_json":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX.json","view_paper":"https://pith.science/paper/LLAVIIL3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14318&json=true","fetch_graph":"https://pith.science/api/pith-number/LLAVIIL3XP4PDQA7KK7EP4CYRX/graph.json","fetch_events":"https://pith.science/api/pith-number/LLAVIIL3XP4PDQA7KK7EP4CYRX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX/action/storage_attestation","attest_author":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX/action/author_attestation","sign_citation":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX/action/citation_signature","submit_replication":"https://pith.science/pith/LLAVIIL3XP4PDQA7KK7EP4CYRX/action/replication_record"}},"created_at":"2026-07-05T10:17:24.302175+00:00","updated_at":"2026-07-05T10:17:24.302175+00:00"}