{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IQZRABNHMOCYID7FYL5Q3GSOJT","short_pith_number":"pith:IQZRABNH","schema_version":"1.0","canonical_sha256":"44331005a76385840fe5c2fb0d9a4e4cda605d7832e72ba19934d140f123041a","source":{"kind":"arxiv","id":"2511.04689","version":3},"attestation_state":"computed","paper":{"title":"Adaptive Testing for LLM Evaluation: A Psychometric Alternative to Static Benchmarks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Nitesh V. Chawla, Peiyu Li, Ronald Metoyer, Si Chen, Ting Hua, Xiuxiu Tang, Ying Cheng","submitted_at":"2025-10-26T03:54:12Z","abstract_excerpt":"Evaluating large language models (LLMs) typically requires thousands of benchmark items, making the process expensive, slow, and increasingly impractical at scale. Existing evaluation protocols rely on average accuracy over fixed item sets, treating all items as equally informative despite substantial variation in difficulty and discrimination. We introduce ATLAS, an adaptive testing framework based on Item Response Theory (IRT) that estimates model ability using Fisher information-guided item selection. ATLAS reduces the number of required items by up to 90% while maintaining measurement prec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2511.04689","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-10-26T03:54:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"34ceadfb5663048b1607c9ae8d92d5b3079648338abca90558c37d22a5c62142","abstract_canon_sha256":"21fa915fddb4b83ca8d81fb0893a080040166d87ec626e90f35b4519b198793a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-15T00:21:13.643980Z","signature_b64":"8tD2qqnmhQpJO43zb7+lHRAEYIyODBXTfy8Ddd5Cz9/DaGmfeyCkoPD4ykQTsrhlxTchBh3q46lSsCYGmDWyCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"44331005a76385840fe5c2fb0d9a4e4cda605d7832e72ba19934d140f123041a","last_reissued_at":"2026-07-15T00:21:13.643010Z","signature_status":"signed_v1","first_computed_at":"2026-07-15T00:21:13.643010Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Testing for LLM Evaluation: A Psychometric Alternative to Static Benchmarks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Nitesh V. Chawla, Peiyu Li, Ronald Metoyer, Si Chen, Ting Hua, Xiuxiu Tang, Ying Cheng","submitted_at":"2025-10-26T03:54:12Z","abstract_excerpt":"Evaluating large language models (LLMs) typically requires thousands of benchmark items, making the process expensive, slow, and increasingly impractical at scale. Existing evaluation protocols rely on average accuracy over fixed item sets, treating all items as equally informative despite substantial variation in difficulty and discrimination. We introduce ATLAS, an adaptive testing framework based on Item Response Theory (IRT) that estimates model ability using Fisher information-guided item selection. ATLAS reduces the number of required items by up to 90% while maintaining measurement prec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2511.04689","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2511.04689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2511.04689","created_at":"2026-07-15T00:21:13.643471+00:00"},{"alias_kind":"arxiv_version","alias_value":"2511.04689v3","created_at":"2026-07-15T00:21:13.643471+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2511.04689","created_at":"2026-07-15T00:21:13.643471+00:00"},{"alias_kind":"pith_short_12","alias_value":"IQZRABNHMOCY","created_at":"2026-07-15T00:21:13.643471+00:00"},{"alias_kind":"pith_short_16","alias_value":"IQZRABNHMOCYID7F","created_at":"2026-07-15T00:21:13.643471+00:00"},{"alias_kind":"pith_short_8","alias_value":"IQZRABNH","created_at":"2026-07-15T00:21:13.643471+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":9,"sample":[{"citing_arxiv_id":"2606.26429","citing_title":"DualEval: Joint Model-Item Calibration for Unified LLM Evaluation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.14516","citing_title":"Every Eval Ever: A Unifying Schema and Community Repository for AI Evaluation Results","ref_index":55,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07422","citing_title":"The Masked Advantage: Uncovering Local-Language Access to Cultural Knowledge in LLMs","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09878","citing_title":"FailureScope: Cross-Regime Behavioral Diagnosis of Language Model Weaknesses","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17110","citing_title":"Capturing LLM Capabilities via Evidence-Calibrated Query Clustering","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30504","citing_title":"Auditing LLM Benchmarks with Item Response Theory","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17110","citing_title":"Capturing LLM Capabilities via Evidence-Calibrated Query Clustering","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2604.12843","citing_title":"Growing Pains: Extensible and Efficient LLM Benchmarking Via Fixed Parameter Calibration","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT","json":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT.json","graph_json":"https://pith.science/api/pith-number/IQZRABNHMOCYID7FYL5Q3GSOJT/graph.json","events_json":"https://pith.science/api/pith-number/IQZRABNHMOCYID7FYL5Q3GSOJT/events.json","paper":"https://pith.science/paper/IQZRABNH"},"agent_actions":{"view_html":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT","download_json":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT.json","view_paper":"https://pith.science/paper/IQZRABNH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2511.04689&json=true","fetch_graph":"https://pith.science/api/pith-number/IQZRABNHMOCYID7FYL5Q3GSOJT/graph.json","fetch_events":"https://pith.science/api/pith-number/IQZRABNHMOCYID7FYL5Q3GSOJT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT/action/storage_attestation","attest_author":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT/action/author_attestation","sign_citation":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT/action/citation_signature","submit_replication":"https://pith.science/pith/IQZRABNHMOCYID7FYL5Q3GSOJT/action/replication_record"}},"created_at":"2026-07-15T00:21:13.643471+00:00","updated_at":"2026-07-15T00:21:13.643471+00:00"}