{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6MI3RH75JLOBNOZRUDU7TCSO27","short_pith_number":"pith:6MI3RH75","schema_version":"1.0","canonical_sha256":"f311b89ffd4adc16bb31a0e9f98a4ed7fea9f21fe23ccee14852c133fb0569ca","source":{"kind":"arxiv","id":"2412.11831","version":2},"attestation_state":"computed","paper":{"title":"Are You Doubtful? Oh, It Might Be Difficult Then! Exploring the Use of Model Uncertainty for Question Difficulty Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hedderik van Rijn, Leonidas Zotos, Malvina Nissim","submitted_at":"2024-12-16T14:55:09Z","abstract_excerpt":"In an educational setting, an estimate of the difficulty of multiple-choice questions (MCQs), a commonly used strategy to assess learning progress, constitutes very useful information for both teachers and students. Since human assessment is costly from multiple points of view, automatic approaches to MCQ item difficulty estimation are investigated, yielding however mixed success until now. Our approach to this problem takes a different angle from previous work: asking various Large Language Models to tackle the questions included in three different MCQ datasets, we leverage model uncertainty "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.11831","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-16T14:55:09Z","cross_cats_sorted":[],"title_canon_sha256":"aeb0d98dc4b079ef81efefe735bebd65b56d76d9397432b1fb24f004bbe48d79","abstract_canon_sha256":"583fa1b3091576afc63cd49cafefa6a76ab35fa3db07568d15f69a0fad1ece2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:46.114859Z","signature_b64":"c3kyWWGqD9WJ2WbXNt3ZzZJfuFMR1flc07ubBRDothr6YiynAJwSkJcvvbqJF/vsbjgTgu5IKLmek/POAq7ZAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f311b89ffd4adc16bb31a0e9f98a4ed7fea9f21fe23ccee14852c133fb0569ca","last_reissued_at":"2026-07-05T10:50:46.114327Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:46.114327Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are You Doubtful? Oh, It Might Be Difficult Then! Exploring the Use of Model Uncertainty for Question Difficulty Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hedderik van Rijn, Leonidas Zotos, Malvina Nissim","submitted_at":"2024-12-16T14:55:09Z","abstract_excerpt":"In an educational setting, an estimate of the difficulty of multiple-choice questions (MCQs), a commonly used strategy to assess learning progress, constitutes very useful information for both teachers and students. Since human assessment is costly from multiple points of view, automatic approaches to MCQ item difficulty estimation are investigated, yielding however mixed success until now. Our approach to this problem takes a different angle from previous work: asking various Large Language Models to tackle the questions included in three different MCQ datasets, we leverage model uncertainty "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.11831","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.11831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.11831","created_at":"2026-07-05T10:50:46.114387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.11831v2","created_at":"2026-07-05T10:50:46.114387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.11831","created_at":"2026-07-05T10:50:46.114387+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MI3RH75JLOB","created_at":"2026-07-05T10:50:46.114387+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MI3RH75JLOBNOZR","created_at":"2026-07-05T10:50:46.114387+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MI3RH75","created_at":"2026-07-05T10:50:46.114387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18709","citing_title":"LLMs Struggle to Measure What Distinguishes Students of Different Proficiency Levels: A Study of Item Discrimination in Reading Comprehension Assessment","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28186","citing_title":"Cognitive Episodes in LLM Reasoning Traces Enable Interpretable Human Item Difficulty Prediction","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27","json":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27.json","graph_json":"https://pith.science/api/pith-number/6MI3RH75JLOBNOZRUDU7TCSO27/graph.json","events_json":"https://pith.science/api/pith-number/6MI3RH75JLOBNOZRUDU7TCSO27/events.json","paper":"https://pith.science/paper/6MI3RH75"},"agent_actions":{"view_html":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27","download_json":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27.json","view_paper":"https://pith.science/paper/6MI3RH75","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.11831&json=true","fetch_graph":"https://pith.science/api/pith-number/6MI3RH75JLOBNOZRUDU7TCSO27/graph.json","fetch_events":"https://pith.science/api/pith-number/6MI3RH75JLOBNOZRUDU7TCSO27/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27/action/storage_attestation","attest_author":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27/action/author_attestation","sign_citation":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27/action/citation_signature","submit_replication":"https://pith.science/pith/6MI3RH75JLOBNOZRUDU7TCSO27/action/replication_record"}},"created_at":"2026-07-05T10:50:46.114387+00:00","updated_at":"2026-07-05T10:50:46.114387+00:00"}