{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CMTADOIODTHFI4XHQRU35DAEDF","short_pith_number":"pith:CMTADOIO","schema_version":"1.0","canonical_sha256":"132601b90e1cce5472e78469be8c04194bc0b4de91ec2703c10af7671460034b","source":{"kind":"arxiv","id":"2507.02856","version":1},"attestation_state":"computed","paper":{"title":"Answer Matching Outperforms Multiple Choice for Language Model Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ameya Prabhu, Jonas Geiping, Moritz Hardt, Nikhil Chandak, Shashwat Goel","submitted_at":"2025-07-03T17:59:02Z","abstract_excerpt":"Multiple choice benchmarks have long been the workhorse of language model evaluation because grading multiple choice is objective and easy to automate. However, we show multiple choice questions from popular benchmarks can often be answered without even seeing the question. These shortcuts arise from a fundamental limitation of discriminative evaluation not shared by evaluations of the model's free-form, generative answers. Until recently, there appeared to be no viable, scalable alternative to multiple choice--but, we show that this has changed. We consider generative evaluation via what we c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.02856","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-03T17:59:02Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e484311a96b94df1e142d2fa6299c7e3db7a48ba4f8916fcadd294968e161d1f","abstract_canon_sha256":"0dbc0c111c2d2f9f417aeb73bd53d3bf99b10b87ed364ba6348c166b647695c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:35.401030Z","signature_b64":"jcTCpT2LMyeUUFdgmanvYgcUdgXM1V5F936xUx5GmaXO3+Q3vlHM/Lqy5gjlZJT0Ma4aIcMgnak14L/q8vT5Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"132601b90e1cce5472e78469be8c04194bc0b4de91ec2703c10af7671460034b","last_reissued_at":"2026-07-05T11:31:35.400545Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:35.400545Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Answer Matching Outperforms Multiple Choice for Language Model Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ameya Prabhu, Jonas Geiping, Moritz Hardt, Nikhil Chandak, Shashwat Goel","submitted_at":"2025-07-03T17:59:02Z","abstract_excerpt":"Multiple choice benchmarks have long been the workhorse of language model evaluation because grading multiple choice is objective and easy to automate. However, we show multiple choice questions from popular benchmarks can often be answered without even seeing the question. These shortcuts arise from a fundamental limitation of discriminative evaluation not shared by evaluations of the model's free-form, generative answers. Until recently, there appeared to be no viable, scalable alternative to multiple choice--but, we show that this has changed. We consider generative evaluation via what we c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.02856","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.02856/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.02856","created_at":"2026-07-05T11:31:35.400605+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.02856v1","created_at":"2026-07-05T11:31:35.400605+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.02856","created_at":"2026-07-05T11:31:35.400605+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMTADOIODTHF","created_at":"2026-07-05T11:31:35.400605+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMTADOIODTHFI4XH","created_at":"2026-07-05T11:31:35.400605+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMTADOIO","created_at":"2026-07-05T11:31:35.400605+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24819","citing_title":"HelpBench: Assessing the Ability of LLMs to Provide Privacy, Safety, and Security Advice","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20900","citing_title":"Storyline Trees: Hierarchical Representations for Long-Form Narratives","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11643","citing_title":"Improving Cross-Format Robustness in Language Models with Multi-Format Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10657","citing_title":"Are We Evaluating Knowledge or Phrasing? Mitigating MCQA Sensitivity with ParaEval","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09409","citing_title":"Correct Looks Better: Pairwise Comparisons Reveal Accuracy Rankings","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.26522","citing_title":"Entropy After </Think> for reasoning model early exiting","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2510.24328","citing_title":"Beyond MCQ: An Open-Ended Arabic Cultural QA Benchmark with Dialect Variants","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF","json":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF.json","graph_json":"https://pith.science/api/pith-number/CMTADOIODTHFI4XHQRU35DAEDF/graph.json","events_json":"https://pith.science/api/pith-number/CMTADOIODTHFI4XHQRU35DAEDF/events.json","paper":"https://pith.science/paper/CMTADOIO"},"agent_actions":{"view_html":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF","download_json":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF.json","view_paper":"https://pith.science/paper/CMTADOIO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.02856&json=true","fetch_graph":"https://pith.science/api/pith-number/CMTADOIODTHFI4XHQRU35DAEDF/graph.json","fetch_events":"https://pith.science/api/pith-number/CMTADOIODTHFI4XHQRU35DAEDF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF/action/storage_attestation","attest_author":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF/action/author_attestation","sign_citation":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF/action/citation_signature","submit_replication":"https://pith.science/pith/CMTADOIODTHFI4XHQRU35DAEDF/action/replication_record"}},"created_at":"2026-07-05T11:31:35.400605+00:00","updated_at":"2026-07-05T11:31:35.400605+00:00"}