{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PWQEXTIDW4ORDW57AXAE6VYBXN","short_pith_number":"pith:PWQEXTID","schema_version":"1.0","canonical_sha256":"7da04bcd03b71d11dbbf05c04f5701bb5090f39b430ca91a8249177045ec4a5f","source":{"kind":"arxiv","id":"2404.10704","version":1},"attestation_state":"computed","paper":{"title":"Question Difficulty Ranking for Multiple-Choice Reading Comprehension","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Mark Gales, Vatsal Raina","submitted_at":"2024-04-16T16:23:10Z","abstract_excerpt":"Multiple-choice (MC) tests are an efficient method to assess English learners. It is useful for test creators to rank candidate MC questions by difficulty during exam curation. Typically, the difficulty is determined by having human test takers trial the questions in a pretesting stage. However, this is expensive and not scalable. Therefore, we explore automated approaches to rank MC questions by difficulty. However, there is limited data for explicit training of a system for difficulty scores. Hence, we compare task transfer and zero-shot approaches: task transfer adapts level classification "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.10704","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-16T16:23:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8aee54ce1a3299810c209cc8e90030313acc4a5d8a25b37db5f4400025c8d919","abstract_canon_sha256":"a7b4052d65a52d9dfb0d33ea175a383a5b1f40f66e8654d58da6dba6a6ac2cb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:43.937524Z","signature_b64":"FGmkjcOQTjIOihmSLGc51x/6EGF7dC6Nvkfuj2Lt1ufXmJPFrVssjNe6iqnIKmUB0gw470VvXlMcmt53dpZ+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7da04bcd03b71d11dbbf05c04f5701bb5090f39b430ca91a8249177045ec4a5f","last_reissued_at":"2026-07-05T08:08:43.936712Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:43.936712Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Question Difficulty Ranking for Multiple-Choice Reading Comprehension","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Mark Gales, Vatsal Raina","submitted_at":"2024-04-16T16:23:10Z","abstract_excerpt":"Multiple-choice (MC) tests are an efficient method to assess English learners. It is useful for test creators to rank candidate MC questions by difficulty during exam curation. Typically, the difficulty is determined by having human test takers trial the questions in a pretesting stage. However, this is expensive and not scalable. Therefore, we explore automated approaches to rank MC questions by difficulty. However, there is limited data for explicit training of a system for difficulty scores. Hence, we compare task transfer and zero-shot approaches: task transfer adapts level classification "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.10704","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.10704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.10704","created_at":"2026-07-05T08:08:43.936764+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.10704v1","created_at":"2026-07-05T08:08:43.936764+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.10704","created_at":"2026-07-05T08:08:43.936764+00:00"},{"alias_kind":"pith_short_12","alias_value":"PWQEXTIDW4OR","created_at":"2026-07-05T08:08:43.936764+00:00"},{"alias_kind":"pith_short_16","alias_value":"PWQEXTIDW4ORDW57","created_at":"2026-07-05T08:08:43.936764+00:00"},{"alias_kind":"pith_short_8","alias_value":"PWQEXTID","created_at":"2026-07-05T08:08:43.936764+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19316","citing_title":"A Multi-Agent Framework for Feature-Constrained Difficulty Control in Reading Comprehension Item Generation","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN","json":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN.json","graph_json":"https://pith.science/api/pith-number/PWQEXTIDW4ORDW57AXAE6VYBXN/graph.json","events_json":"https://pith.science/api/pith-number/PWQEXTIDW4ORDW57AXAE6VYBXN/events.json","paper":"https://pith.science/paper/PWQEXTID"},"agent_actions":{"view_html":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN","download_json":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN.json","view_paper":"https://pith.science/paper/PWQEXTID","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.10704&json=true","fetch_graph":"https://pith.science/api/pith-number/PWQEXTIDW4ORDW57AXAE6VYBXN/graph.json","fetch_events":"https://pith.science/api/pith-number/PWQEXTIDW4ORDW57AXAE6VYBXN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN/action/storage_attestation","attest_author":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN/action/author_attestation","sign_citation":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN/action/citation_signature","submit_replication":"https://pith.science/pith/PWQEXTIDW4ORDW57AXAE6VYBXN/action/replication_record"}},"created_at":"2026-07-05T08:08:43.936764+00:00","updated_at":"2026-07-05T08:08:43.936764+00:00"}