{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:R3QQITW5EUT5RT7R6RGCJQ66GI","short_pith_number":"pith:R3QQITW5","schema_version":"1.0","canonical_sha256":"8ee1044edd2527d8cff1f44c24c3de32216ad736aa07618eae2f956f526b6253","source":{"kind":"arxiv","id":"2506.09796","version":1},"attestation_state":"computed","paper":{"title":"Do LLMs Give Psychometrically Plausible Responses in Educational Assessments?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andreas S\\\"auberli, Barbara Plank, Diego Frassinelli","submitted_at":"2025-06-11T14:41:10Z","abstract_excerpt":"Knowing how test takers answer items in educational assessments is essential for test development, to evaluate item quality, and to improve test validity. However, this process usually requires extensive pilot studies with human participants. If large language models (LLMs) exhibit human-like response behavior to test items, this could open up the possibility of using them as pilot participants to accelerate test development. In this paper, we evaluate the human-likeness or psychometric plausibility of responses from 18 instruction-tuned LLMs with two publicly available datasets of multiple-ch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09796","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-11T14:41:10Z","cross_cats_sorted":[],"title_canon_sha256":"7272f12657f3823d69d0d35e69e9833068c71562eb1f98779a722d3bd0c6b66e","abstract_canon_sha256":"2f85f559af3a6d0e56668f10a6f1c3e9fcff2672547cd0699735c154971c45f9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:57.126007Z","signature_b64":"wDRbL/4SgY6AnvDdy11hMCtdoJjqu4WRjfCFyDMEHFuplBsDBOj0SVKI9i4EWmQ0liH1UFMtGo5mO9oBlfaqAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ee1044edd2527d8cff1f44c24c3de32216ad736aa07618eae2f956f526b6253","last_reissued_at":"2026-07-05T11:19:57.125298Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:57.125298Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do LLMs Give Psychometrically Plausible Responses in Educational Assessments?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andreas S\\\"auberli, Barbara Plank, Diego Frassinelli","submitted_at":"2025-06-11T14:41:10Z","abstract_excerpt":"Knowing how test takers answer items in educational assessments is essential for test development, to evaluate item quality, and to improve test validity. However, this process usually requires extensive pilot studies with human participants. If large language models (LLMs) exhibit human-like response behavior to test items, this could open up the possibility of using them as pilot participants to accelerate test development. In this paper, we evaluate the human-likeness or psychometric plausibility of responses from 18 instruction-tuned LLMs with two publicly available datasets of multiple-ch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09796","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09796/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09796","created_at":"2026-07-05T11:19:57.125401+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09796v1","created_at":"2026-07-05T11:19:57.125401+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09796","created_at":"2026-07-05T11:19:57.125401+00:00"},{"alias_kind":"pith_short_12","alias_value":"R3QQITW5EUT5","created_at":"2026-07-05T11:19:57.125401+00:00"},{"alias_kind":"pith_short_16","alias_value":"R3QQITW5EUT5RT7R","created_at":"2026-07-05T11:19:57.125401+00:00"},{"alias_kind":"pith_short_8","alias_value":"R3QQITW5","created_at":"2026-07-05T11:19:57.125401+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18709","citing_title":"LLMs Struggle to Measure What Distinguishes Students of Different Proficiency Levels: A Study of Item Discrimination in Reading Comprehension Assessment","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI","json":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI.json","graph_json":"https://pith.science/api/pith-number/R3QQITW5EUT5RT7R6RGCJQ66GI/graph.json","events_json":"https://pith.science/api/pith-number/R3QQITW5EUT5RT7R6RGCJQ66GI/events.json","paper":"https://pith.science/paper/R3QQITW5"},"agent_actions":{"view_html":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI","download_json":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI.json","view_paper":"https://pith.science/paper/R3QQITW5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09796&json=true","fetch_graph":"https://pith.science/api/pith-number/R3QQITW5EUT5RT7R6RGCJQ66GI/graph.json","fetch_events":"https://pith.science/api/pith-number/R3QQITW5EUT5RT7R6RGCJQ66GI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI/action/storage_attestation","attest_author":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI/action/author_attestation","sign_citation":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI/action/citation_signature","submit_replication":"https://pith.science/pith/R3QQITW5EUT5RT7R6RGCJQ66GI/action/replication_record"}},"created_at":"2026-07-05T11:19:57.125401+00:00","updated_at":"2026-07-05T11:19:57.125401+00:00"}