{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DHJJXNWH6MXZHVQIJTAEFCNFXB","short_pith_number":"pith:DHJJXNWH","schema_version":"1.0","canonical_sha256":"19d29bb6c7f32f93d6084cc04289a5b8722c76360d3270faa33fe061cb1573ed","source":{"kind":"arxiv","id":"2509.11106","version":1},"attestation_state":"computed","paper":{"title":"Fluid Language Model Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chun Wang, David Heineman, Hannaneh Hajishirzi, Ian Magnusson, Jesse Dodge, Kyle Lo, Maarten Sap, Noah A. Smith, Pang Wei Koh, Valentin Hofmann","submitted_at":"2025-09-14T05:49:42Z","abstract_excerpt":"Language model (LM) benchmarking faces several challenges: comprehensive evaluations are costly, benchmarks often fail to measure the intended capabilities, and evaluation quality can degrade due to labeling errors and benchmark saturation. Although various strategies have been proposed to mitigate these issues, they tend to address individual aspects in isolation, neglecting broader questions about overall evaluation quality. Here, we introduce Fluid Benchmarking, a new evaluation approach that advances LM benchmarking across multiple dimensions. Inspired by psychometrics, Fluid Benchmarking "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.11106","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-14T05:49:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"18b55c17488cc5054b9dd4d5d1193ef6db7b9f9a9673c280455f9cadbafeaa5a","abstract_canon_sha256":"5eca93cebaa40444044cdcb33e16c26e1924428ecc71d4984a1b8fc7068f931b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:33.655690Z","signature_b64":"kj7egW+zPxtGFEtvOO3ZEJvV/XjsfgdL7r1ywzqQmJwiQK0JUjkGkDEGct1rFUOAAnyDjR7590eoAV1q+51kCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19d29bb6c7f32f93d6084cc04289a5b8722c76360d3270faa33fe061cb1573ed","last_reissued_at":"2026-07-05T12:11:33.655191Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:33.655191Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fluid Language Model Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chun Wang, David Heineman, Hannaneh Hajishirzi, Ian Magnusson, Jesse Dodge, Kyle Lo, Maarten Sap, Noah A. Smith, Pang Wei Koh, Valentin Hofmann","submitted_at":"2025-09-14T05:49:42Z","abstract_excerpt":"Language model (LM) benchmarking faces several challenges: comprehensive evaluations are costly, benchmarks often fail to measure the intended capabilities, and evaluation quality can degrade due to labeling errors and benchmark saturation. Although various strategies have been proposed to mitigate these issues, they tend to address individual aspects in isolation, neglecting broader questions about overall evaluation quality. Here, we introduce Fluid Benchmarking, a new evaluation approach that advances LM benchmarking across multiple dimensions. Inspired by psychometrics, Fluid Benchmarking "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.11106","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.11106/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.11106","created_at":"2026-07-05T12:11:33.655247+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.11106v1","created_at":"2026-07-05T12:11:33.655247+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.11106","created_at":"2026-07-05T12:11:33.655247+00:00"},{"alias_kind":"pith_short_12","alias_value":"DHJJXNWH6MXZ","created_at":"2026-07-05T12:11:33.655247+00:00"},{"alias_kind":"pith_short_16","alias_value":"DHJJXNWH6MXZHVQI","created_at":"2026-07-05T12:11:33.655247+00:00"},{"alias_kind":"pith_short_8","alias_value":"DHJJXNWH","created_at":"2026-07-05T12:11:33.655247+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07616","citing_title":"Item Response Scaling Laws: A Measurement Theory Approach for Efficient and Generalizable Neural Scaling Estimation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11209","citing_title":"Measuring Five-Nines Reliability: Sample-Efficient LLM Evaluation in Saturated Benchmarks","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12843","citing_title":"Growing Pains: Extensible and Efficient LLM Benchmarking Via Fixed Parameter Calibration","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB","json":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB.json","graph_json":"https://pith.science/api/pith-number/DHJJXNWH6MXZHVQIJTAEFCNFXB/graph.json","events_json":"https://pith.science/api/pith-number/DHJJXNWH6MXZHVQIJTAEFCNFXB/events.json","paper":"https://pith.science/paper/DHJJXNWH"},"agent_actions":{"view_html":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB","download_json":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB.json","view_paper":"https://pith.science/paper/DHJJXNWH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.11106&json=true","fetch_graph":"https://pith.science/api/pith-number/DHJJXNWH6MXZHVQIJTAEFCNFXB/graph.json","fetch_events":"https://pith.science/api/pith-number/DHJJXNWH6MXZHVQIJTAEFCNFXB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB/action/storage_attestation","attest_author":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB/action/author_attestation","sign_citation":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB/action/citation_signature","submit_replication":"https://pith.science/pith/DHJJXNWH6MXZHVQIJTAEFCNFXB/action/replication_record"}},"created_at":"2026-07-05T12:11:33.655247+00:00","updated_at":"2026-07-05T12:11:33.655247+00:00"}