{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6HD56LLJ2Y3FIXPM6QM2IMWXLG","short_pith_number":"pith:6HD56LLJ","schema_version":"1.0","canonical_sha256":"f1c7df2d69d636545decf419a432d7598b8570f7ee9786501aaf00d9d61ea9d4","source":{"kind":"arxiv","id":"2401.07955","version":2},"attestation_state":"computed","paper":{"title":"A Study on Large Language Models' Limitations in Multiple-Choice Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aisha Khatun, Daniel G. Brown","submitted_at":"2024-01-15T20:42:16Z","abstract_excerpt":"The widespread adoption of Large Language Models (LLMs) has become commonplace, particularly with the emergence of open-source models. More importantly, smaller models are well-suited for integration into consumer devices and are frequently employed either as standalone solutions or as subroutines in various AI tasks. Despite their ubiquitous use, there is no systematic analysis of their specific capabilities and limitations. In this study, we tackle one of the most widely used tasks - answering Multiple Choice Question (MCQ). We analyze 26 small open-source models and find that 65% of the mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.07955","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-15T20:42:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d979b387cc80c2eda5f77bf968e685fe779bc5b0495a11cf5caf6e4bd577bd9a","abstract_canon_sha256":"3d44733e050ef9635e73f981c12883005a271cb49bb5cc773dbde8ed19d834be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:32.969820Z","signature_b64":"0IQQV+Bzht6dCCT/kJG+crxkuey4wjMoEWvqOm4Yy1/hIvQ11IZN/sq0IlsfAqd2R69BrgRYFAlVlK28DzOGCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1c7df2d69d636545decf419a432d7598b8570f7ee9786501aaf00d9d61ea9d4","last_reissued_at":"2026-07-05T08:55:32.969433Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:32.969433Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Study on Large Language Models' Limitations in Multiple-Choice Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aisha Khatun, Daniel G. Brown","submitted_at":"2024-01-15T20:42:16Z","abstract_excerpt":"The widespread adoption of Large Language Models (LLMs) has become commonplace, particularly with the emergence of open-source models. More importantly, smaller models are well-suited for integration into consumer devices and are frequently employed either as standalone solutions or as subroutines in various AI tasks. Despite their ubiquitous use, there is no systematic analysis of their specific capabilities and limitations. In this study, we tackle one of the most widely used tasks - answering Multiple Choice Question (MCQ). We analyze 26 small open-source models and find that 65% of the mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.07955","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.07955/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.07955","created_at":"2026-07-05T08:55:32.969488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.07955v2","created_at":"2026-07-05T08:55:32.969488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.07955","created_at":"2026-07-05T08:55:32.969488+00:00"},{"alias_kind":"pith_short_12","alias_value":"6HD56LLJ2Y3F","created_at":"2026-07-05T08:55:32.969488+00:00"},{"alias_kind":"pith_short_16","alias_value":"6HD56LLJ2Y3FIXPM","created_at":"2026-07-05T08:55:32.969488+00:00"},{"alias_kind":"pith_short_8","alias_value":"6HD56LLJ","created_at":"2026-07-05T08:55:32.969488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17467","citing_title":"Probing Vision-Language Understanding through the Visual Entailment Task: promises and pitfalls","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG","json":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG.json","graph_json":"https://pith.science/api/pith-number/6HD56LLJ2Y3FIXPM6QM2IMWXLG/graph.json","events_json":"https://pith.science/api/pith-number/6HD56LLJ2Y3FIXPM6QM2IMWXLG/events.json","paper":"https://pith.science/paper/6HD56LLJ"},"agent_actions":{"view_html":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG","download_json":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG.json","view_paper":"https://pith.science/paper/6HD56LLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.07955&json=true","fetch_graph":"https://pith.science/api/pith-number/6HD56LLJ2Y3FIXPM6QM2IMWXLG/graph.json","fetch_events":"https://pith.science/api/pith-number/6HD56LLJ2Y3FIXPM6QM2IMWXLG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG/action/storage_attestation","attest_author":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG/action/author_attestation","sign_citation":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG/action/citation_signature","submit_replication":"https://pith.science/pith/6HD56LLJ2Y3FIXPM6QM2IMWXLG/action/replication_record"}},"created_at":"2026-07-05T08:55:32.969488+00:00","updated_at":"2026-07-05T08:55:32.969488+00:00"}