{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:INMB5WAMSFTB2KGTYIFENDC223","short_pith_number":"pith:INMB5WAM","schema_version":"1.0","canonical_sha256":"43581ed80c91661d28d3c20a468c5ad6c6ce305891bcf85cff7a243e6706a83c","source":{"kind":"arxiv","id":"2404.08382","version":2},"attestation_state":"computed","paper":{"title":"Look at the Text: Instruction-Tuned Language Models are More Robust Multiple Choice Selectors than You Think","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Bolei Ma, Chengzhi Hu, Paul R\\\"ottger, Xinpeng Wang","submitted_at":"2024-04-12T10:36:15Z","abstract_excerpt":"Multiple choice questions (MCQs) are commonly used to evaluate the capabilities of large language models (LLMs). One common way to evaluate the model response is to rank the candidate answers based on the log probability of the first token prediction. An alternative way is to examine the text output. Prior work has shown that first token probabilities lack robustness to changes in MCQ phrasing, and that first token probabilities do not match text answers for instruction-tuned models. Therefore, in this paper, we investigate the robustness of text answers. We show that the text answers are more"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.08382","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-12T10:36:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5164ab7f590806223cbc098c3e0c0b1b6e5d60c9b43d5799095a89c8cb64cda4","abstract_canon_sha256":"2f5ee27bf08da52557ff989ddad06929e69a2d3dbbfcf6e37e8fea0ff2c74a9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:02.249304Z","signature_b64":"bzytkEFxUL4DiIFi53y2dE72oJz+ddPQ3WW0XZg2SH1UcMGmilpdtz+qJY90XmuGonYJjXzXlRPHj/GlSmpxCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43581ed80c91661d28d3c20a468c5ad6c6ce305891bcf85cff7a243e6706a83c","last_reissued_at":"2026-07-05T08:57:02.248896Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:02.248896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look at the Text: Instruction-Tuned Language Models are More Robust Multiple Choice Selectors than You Think","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Bolei Ma, Chengzhi Hu, Paul R\\\"ottger, Xinpeng Wang","submitted_at":"2024-04-12T10:36:15Z","abstract_excerpt":"Multiple choice questions (MCQs) are commonly used to evaluate the capabilities of large language models (LLMs). One common way to evaluate the model response is to rank the candidate answers based on the log probability of the first token prediction. An alternative way is to examine the text output. Prior work has shown that first token probabilities lack robustness to changes in MCQ phrasing, and that first token probabilities do not match text answers for instruction-tuned models. Therefore, in this paper, we investigate the robustness of text answers. We show that the text answers are more"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08382","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.08382/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.08382","created_at":"2026-07-05T08:57:02.248951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.08382v2","created_at":"2026-07-05T08:57:02.248951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08382","created_at":"2026-07-05T08:57:02.248951+00:00"},{"alias_kind":"pith_short_12","alias_value":"INMB5WAMSFTB","created_at":"2026-07-05T08:57:02.248951+00:00"},{"alias_kind":"pith_short_16","alias_value":"INMB5WAMSFTB2KGT","created_at":"2026-07-05T08:57:02.248951+00:00"},{"alias_kind":"pith_short_8","alias_value":"INMB5WAM","created_at":"2026-07-05T08:57:02.248951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.18182","citing_title":"SCOPE: Stochastic and Counterbiased Option Placement for Evaluating Large Language Models","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223","json":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223.json","graph_json":"https://pith.science/api/pith-number/INMB5WAMSFTB2KGTYIFENDC223/graph.json","events_json":"https://pith.science/api/pith-number/INMB5WAMSFTB2KGTYIFENDC223/events.json","paper":"https://pith.science/paper/INMB5WAM"},"agent_actions":{"view_html":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223","download_json":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223.json","view_paper":"https://pith.science/paper/INMB5WAM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.08382&json=true","fetch_graph":"https://pith.science/api/pith-number/INMB5WAMSFTB2KGTYIFENDC223/graph.json","fetch_events":"https://pith.science/api/pith-number/INMB5WAMSFTB2KGTYIFENDC223/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223/action/timestamp_anchor","attest_storage":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223/action/storage_attestation","attest_author":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223/action/author_attestation","sign_citation":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223/action/citation_signature","submit_replication":"https://pith.science/pith/INMB5WAMSFTB2KGTYIFENDC223/action/replication_record"}},"created_at":"2026-07-05T08:57:02.248951+00:00","updated_at":"2026-07-05T08:57:02.248951+00:00"}