{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OYCRJRJMUX3YEEKCJR63L62JGX","short_pith_number":"pith:OYCRJRJM","schema_version":"1.0","canonical_sha256":"760514c52ca5f78211424c7db5fb4935c1c9579f146357a37f8a011533dfce33","source":{"kind":"arxiv","id":"2406.07545","version":1},"attestation_state":"computed","paper":{"title":"Open-LLM-Leaderboard: From Multi-choice to Open-style Questions for LLMs Evaluation, Benchmark, and Arena","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aidar Myrzakhan, Sondos Mahmoud Bsharat, Zhiqiang Shen","submitted_at":"2024-06-11T17:59:47Z","abstract_excerpt":"Multiple-choice questions (MCQ) are frequently used to assess large language models (LLMs). Typically, an LLM is given a question and selects the answer deemed most probable after adjustments for factors like length. Unfortunately, LLMs may inherently favor certain answer choice IDs, such as A/B/C/D, due to inherent biases of priori unbalanced probabilities, influencing the prediction of answers based on these IDs. Previous research has introduced methods to reduce this ''selection bias'' by simply permutating options on a few test samples and applying to new ones. Another problem of MCQ is th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07545","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-11T17:59:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1ea82b2fdb225472bfa1b860da664e0b21386579cec98ab7c0cde21d66bed521","abstract_canon_sha256":"4bb86d1e0fa86c50e114b7d3955957340b28e740954f504a91b747ec6841f22b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:26.853775Z","signature_b64":"88LZ8z0Q8M3sh0Cc/wwJUb57TJnpTwdyMGUXWhZ0PenCXFN1uoGJlWJoMMXGy6qeUE4ZtMMe+HhwZSN64l6wBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"760514c52ca5f78211424c7db5fb4935c1c9579f146357a37f8a011533dfce33","last_reissued_at":"2026-07-05T08:30:26.853255Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:26.853255Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Open-LLM-Leaderboard: From Multi-choice to Open-style Questions for LLMs Evaluation, Benchmark, and Arena","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aidar Myrzakhan, Sondos Mahmoud Bsharat, Zhiqiang Shen","submitted_at":"2024-06-11T17:59:47Z","abstract_excerpt":"Multiple-choice questions (MCQ) are frequently used to assess large language models (LLMs). Typically, an LLM is given a question and selects the answer deemed most probable after adjustments for factors like length. Unfortunately, LLMs may inherently favor certain answer choice IDs, such as A/B/C/D, due to inherent biases of priori unbalanced probabilities, influencing the prediction of answers based on these IDs. Previous research has introduced methods to reduce this ''selection bias'' by simply permutating options on a few test samples and applying to new ones. Another problem of MCQ is th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07545","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07545/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07545","created_at":"2026-07-05T08:30:26.853323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07545v1","created_at":"2026-07-05T08:30:26.853323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07545","created_at":"2026-07-05T08:30:26.853323+00:00"},{"alias_kind":"pith_short_12","alias_value":"OYCRJRJMUX3Y","created_at":"2026-07-05T08:30:26.853323+00:00"},{"alias_kind":"pith_short_16","alias_value":"OYCRJRJMUX3YEEKC","created_at":"2026-07-05T08:30:26.853323+00:00"},{"alias_kind":"pith_short_8","alias_value":"OYCRJRJM","created_at":"2026-07-05T08:30:26.853323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11643","citing_title":"Improving Cross-Format Robustness in Language Models with Multi-Format Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19590","citing_title":"Position: AI Evaluations Should be Grounded on a Theory of Capability","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17196","citing_title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20251","citing_title":"Efficient Evaluation of LLM Performance with Statistical Guarantees","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19786","citing_title":"HumorRank: A Tournament-Based Leaderboard for Evaluating Humor Generation in Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":166,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX","json":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX.json","graph_json":"https://pith.science/api/pith-number/OYCRJRJMUX3YEEKCJR63L62JGX/graph.json","events_json":"https://pith.science/api/pith-number/OYCRJRJMUX3YEEKCJR63L62JGX/events.json","paper":"https://pith.science/paper/OYCRJRJM"},"agent_actions":{"view_html":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX","download_json":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX.json","view_paper":"https://pith.science/paper/OYCRJRJM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07545&json=true","fetch_graph":"https://pith.science/api/pith-number/OYCRJRJMUX3YEEKCJR63L62JGX/graph.json","fetch_events":"https://pith.science/api/pith-number/OYCRJRJMUX3YEEKCJR63L62JGX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX/action/storage_attestation","attest_author":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX/action/author_attestation","sign_citation":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX/action/citation_signature","submit_replication":"https://pith.science/pith/OYCRJRJMUX3YEEKCJR63L62JGX/action/replication_record"}},"created_at":"2026-07-05T08:30:26.853323+00:00","updated_at":"2026-07-05T08:30:26.853323+00:00"}