{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DGNVCWLQQQVDARSBHLHEUBWBI7","short_pith_number":"pith:DGNVCWLQ","schema_version":"1.0","canonical_sha256":"199b515970842a3046413ace4a06c147cd7860878179ebf74bd7b675d27215c0","source":{"kind":"arxiv","id":"2409.08823","version":1},"attestation_state":"computed","paper":{"title":"AutoIRT: Calibrating Item Response Theory Models with Automated Machine Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.AP"],"primary_cat":"cs.LG","authors_text":"Geoff LaFlair, James Sharpnack, Kevin Yancey, Klinton Bicknell, Phoebe Mulcaire","submitted_at":"2024-09-13T13:36:51Z","abstract_excerpt":"Item response theory (IRT) is a class of interpretable factor models that are widely used in computerized adaptive tests (CATs), such as language proficiency tests. Traditionally, these are fit using parametric mixed effects models on the probability of a test taker getting the correct answer to a test item (i.e., question). Neural net extensions of these models, such as BertIRT, require specialized architectures and parameter tuning. We propose a multistage fitting procedure that is compatible with out-of-the-box Automated Machine Learning (AutoML) tools. It is based on a Monte Carlo EM (MCEM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.08823","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-13T13:36:51Z","cross_cats_sorted":["stat.AP"],"title_canon_sha256":"38afc0c6ea56acaaf13246a75a61ddbd6e7c19f13bea6ec870dfdff717276467","abstract_canon_sha256":"679cfe7cdf8365822abe273adea25102b1ec0bfcce283670109628706a4a35e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:06:48.515772Z","signature_b64":"W4ZICO8zMfFaUucAWhM8BtOHEiRzRdGvfBxgiv8B7TTHZxjUlwcM2vrMWa+ZCUtdBXTkg2eeSWAWYDBvfnJzBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"199b515970842a3046413ace4a06c147cd7860878179ebf74bd7b675d27215c0","last_reissued_at":"2026-07-05T09:06:48.515369Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:06:48.515369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoIRT: Calibrating Item Response Theory Models with Automated Machine Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.AP"],"primary_cat":"cs.LG","authors_text":"Geoff LaFlair, James Sharpnack, Kevin Yancey, Klinton Bicknell, Phoebe Mulcaire","submitted_at":"2024-09-13T13:36:51Z","abstract_excerpt":"Item response theory (IRT) is a class of interpretable factor models that are widely used in computerized adaptive tests (CATs), such as language proficiency tests. Traditionally, these are fit using parametric mixed effects models on the probability of a test taker getting the correct answer to a test item (i.e., question). Neural net extensions of these models, such as BertIRT, require specialized architectures and parameter tuning. We propose a multistage fitting procedure that is compatible with out-of-the-box Automated Machine Learning (AutoML) tools. It is based on a Monte Carlo EM (MCEM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.08823","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.08823/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.08823","created_at":"2026-07-05T09:06:48.515428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.08823v1","created_at":"2026-07-05T09:06:48.515428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.08823","created_at":"2026-07-05T09:06:48.515428+00:00"},{"alias_kind":"pith_short_12","alias_value":"DGNVCWLQQQVD","created_at":"2026-07-05T09:06:48.515428+00:00"},{"alias_kind":"pith_short_16","alias_value":"DGNVCWLQQQVDARSB","created_at":"2026-07-05T09:06:48.515428+00:00"},{"alias_kind":"pith_short_8","alias_value":"DGNVCWLQ","created_at":"2026-07-05T09:06:48.515428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06905","citing_title":"Learning Item Embeddings and Hyperparameters for IRT Calibration via Monte Carlo EM","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16991","citing_title":"Response-free item difficulty modelling for multiple-choice items with fine-tuned transformers: Component-wise representation and multi-task learning","ref_index":160,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7","json":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7.json","graph_json":"https://pith.science/api/pith-number/DGNVCWLQQQVDARSBHLHEUBWBI7/graph.json","events_json":"https://pith.science/api/pith-number/DGNVCWLQQQVDARSBHLHEUBWBI7/events.json","paper":"https://pith.science/paper/DGNVCWLQ"},"agent_actions":{"view_html":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7","download_json":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7.json","view_paper":"https://pith.science/paper/DGNVCWLQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.08823&json=true","fetch_graph":"https://pith.science/api/pith-number/DGNVCWLQQQVDARSBHLHEUBWBI7/graph.json","fetch_events":"https://pith.science/api/pith-number/DGNVCWLQQQVDARSBHLHEUBWBI7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7/action/storage_attestation","attest_author":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7/action/author_attestation","sign_citation":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7/action/citation_signature","submit_replication":"https://pith.science/pith/DGNVCWLQQQVDARSBHLHEUBWBI7/action/replication_record"}},"created_at":"2026-07-05T09:06:48.515428+00:00","updated_at":"2026-07-05T09:06:48.515428+00:00"}