{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P6QRWCKKEZFVIRYIDZLZ7G42BP","short_pith_number":"pith:P6QRWCKK","schema_version":"1.0","canonical_sha256":"7fa11b094a264b5447081e579f9b9a0bfa4f6f69d4107595e13b1009b930fbae","source":{"kind":"arxiv","id":"2403.17752","version":3},"attestation_state":"computed","paper":{"title":"Can multiple-choice questions really be useful in detecting the abilities of LLMs?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Liangzhi Li, Noa Garcia, Tong Xiang, Wangyue Li, Wei Deng, Xiao Liu","submitted_at":"2024-03-26T14:43:48Z","abstract_excerpt":"Multiple-choice questions (MCQs) are widely used in the evaluation of large language models (LLMs) due to their simplicity and efficiency. However, there are concerns about whether MCQs can truly measure LLM's capabilities, particularly in knowledge-intensive scenarios where long-form generation (LFG) answers are required. The misalignment between the task and the evaluation method demands a thoughtful analysis of MCQ's efficacy, which we undertake in this paper by evaluating nine LLMs on four question-answering (QA) datasets in two languages: Chinese and English. We identify a significant iss"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.17752","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-26T14:43:48Z","cross_cats_sorted":[],"title_canon_sha256":"8d2fd7c2b91ac483de2db6a460788e147d4d7278e37ea9cd02a59dd3603ad57d","abstract_canon_sha256":"b219455e0836f25a898f42ebea357894705038076a0e624e8a43dabf30a4f1f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:22.747603Z","signature_b64":"74LPiYRL5hjkpAaRQOUH/5j9Lsres08JPz9pw9k7y0B8ROzH0KHRoovxAaowb3bGK4ppsQkpKVJiyPQt4lIuCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fa11b094a264b5447081e579f9b9a0bfa4f6f69d4107595e13b1009b930fbae","last_reissued_at":"2026-07-05T08:22:22.747115Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:22.747115Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can multiple-choice questions really be useful in detecting the abilities of LLMs?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Liangzhi Li, Noa Garcia, Tong Xiang, Wangyue Li, Wei Deng, Xiao Liu","submitted_at":"2024-03-26T14:43:48Z","abstract_excerpt":"Multiple-choice questions (MCQs) are widely used in the evaluation of large language models (LLMs) due to their simplicity and efficiency. However, there are concerns about whether MCQs can truly measure LLM's capabilities, particularly in knowledge-intensive scenarios where long-form generation (LFG) answers are required. The misalignment between the task and the evaluation method demands a thoughtful analysis of MCQ's efficacy, which we undertake in this paper by evaluating nine LLMs on four question-answering (QA) datasets in two languages: Chinese and English. We identify a significant iss"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.17752","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.17752/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.17752","created_at":"2026-07-05T08:22:22.747174+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.17752v3","created_at":"2026-07-05T08:22:22.747174+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.17752","created_at":"2026-07-05T08:22:22.747174+00:00"},{"alias_kind":"pith_short_12","alias_value":"P6QRWCKKEZFV","created_at":"2026-07-05T08:22:22.747174+00:00"},{"alias_kind":"pith_short_16","alias_value":"P6QRWCKKEZFVIRYI","created_at":"2026-07-05T08:22:22.747174+00:00"},{"alias_kind":"pith_short_8","alias_value":"P6QRWCKK","created_at":"2026-07-05T08:22:22.747174+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.00084","citing_title":"Vision-Language and Large Language Model Performance in Gastroenterology: GPT, Claude, Llama, Phi, Mistral, Gemma, and Quantized Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15707","citing_title":"Is Large Language Model Performance on Reasoning Tasks Impacted by Different Ways Questions Are Asked?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12529","citing_title":"BackFlush: Knowledge-Free Backdoor Detection and Elimination with Watermark Preservation in Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01846","citing_title":"Do Large Language Models Plan Answer Positions? Position Bias in Multiple-Choice Question Generation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP","json":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP.json","graph_json":"https://pith.science/api/pith-number/P6QRWCKKEZFVIRYIDZLZ7G42BP/graph.json","events_json":"https://pith.science/api/pith-number/P6QRWCKKEZFVIRYIDZLZ7G42BP/events.json","paper":"https://pith.science/paper/P6QRWCKK"},"agent_actions":{"view_html":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP","download_json":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP.json","view_paper":"https://pith.science/paper/P6QRWCKK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.17752&json=true","fetch_graph":"https://pith.science/api/pith-number/P6QRWCKKEZFVIRYIDZLZ7G42BP/graph.json","fetch_events":"https://pith.science/api/pith-number/P6QRWCKKEZFVIRYIDZLZ7G42BP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP/action/storage_attestation","attest_author":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP/action/author_attestation","sign_citation":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP/action/citation_signature","submit_replication":"https://pith.science/pith/P6QRWCKKEZFVIRYIDZLZ7G42BP/action/replication_record"}},"created_at":"2026-07-05T08:22:22.747174+00:00","updated_at":"2026-07-05T08:22:22.747174+00:00"}