{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MABHCAGCAX3Q5BM5IGFFGWJGU3","short_pith_number":"pith:MABHCAGC","schema_version":"1.0","canonical_sha256":"60027100c205f70e859d418a535926a6e85fa1d3f3fd595544462cef538a8030","source":{"kind":"arxiv","id":"2308.11483","version":1},"attestation_state":"computed","paper":{"title":"Large Language Models Sensitivity to The Order of Options in Multiple-Choice Questions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Estevam Hruschka, Pouya Pezeshkpour","submitted_at":"2023-08-22T14:54:59Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities in various NLP tasks. However, previous works have shown these models are sensitive towards prompt wording, and few-shot demonstrations and their order, posing challenges to fair assessment of these models. As these models become more powerful, it becomes imperative to understand and address these limitations. In this paper, we focus on LLMs robustness on the task of multiple-choice questions -- commonly adopted task to study reasoning and fact-retrieving capability of LLMs. Investigating the sensitivity of LLMs towards the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.11483","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-22T14:54:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"79e25918fdd6f241e3910fe67d60d3ad502ae442f7db62fb5bc5ed5e11b3df1f","abstract_canon_sha256":"179982832a54ab776cf1a1ae1be84a2f3c6d033e4e76d942942bb6328b390095"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:43:35.539143Z","signature_b64":"T09OVcNqQ8J67Swsih0UGefUhSvz1c9+r+PUH5BusjEWY7uxEQB7q271rgVFR0GkHop+4mA2D1UQsi88aKr0DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60027100c205f70e859d418a535926a6e85fa1d3f3fd595544462cef538a8030","last_reissued_at":"2026-07-05T06:43:35.538603Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:43:35.538603Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models Sensitivity to The Order of Options in Multiple-Choice Questions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Estevam Hruschka, Pouya Pezeshkpour","submitted_at":"2023-08-22T14:54:59Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities in various NLP tasks. However, previous works have shown these models are sensitive towards prompt wording, and few-shot demonstrations and their order, posing challenges to fair assessment of these models. As these models become more powerful, it becomes imperative to understand and address these limitations. In this paper, we focus on LLMs robustness on the task of multiple-choice questions -- commonly adopted task to study reasoning and fact-retrieving capability of LLMs. Investigating the sensitivity of LLMs towards the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.11483","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.11483/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.11483","created_at":"2026-07-05T06:43:35.538663+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.11483v1","created_at":"2026-07-05T06:43:35.538663+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.11483","created_at":"2026-07-05T06:43:35.538663+00:00"},{"alias_kind":"pith_short_12","alias_value":"MABHCAGCAX3Q","created_at":"2026-07-05T06:43:35.538663+00:00"},{"alias_kind":"pith_short_16","alias_value":"MABHCAGCAX3Q5BM5","created_at":"2026-07-05T06:43:35.538663+00:00"},{"alias_kind":"pith_short_8","alias_value":"MABHCAGC","created_at":"2026-07-05T06:43:35.538663+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08065","citing_title":"When LLMs Agree, Are They Right? Auditing Self-Consistency and Cross-Model Agreement as Confidence Signals","ref_index":51,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12250","citing_title":"Reassessing High-Performing LLMs on Polish Medical Exams: True Competence or Bias-Driven Performance?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04273","citing_title":"Human agency in initial human-AI proof formalization workflows","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13076","citing_title":"LLM Evaluators Recognize and Favor Their Own Generations","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14092","citing_title":"Fragile Preferences: A Deep Dive Into Order Effects in Large Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12917","citing_title":"Training Language Models to Self-Correct via Reinforcement Learning","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2309.00267","citing_title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04944","citing_title":"Inclusion-of-Thoughts: Mitigating Preference Instability via Purifying the Decision Space","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27249","citing_title":"Instruction Complexity Induces Positional Collapse in Adversarial LLM Evaluation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":184,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3","json":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3.json","graph_json":"https://pith.science/api/pith-number/MABHCAGCAX3Q5BM5IGFFGWJGU3/graph.json","events_json":"https://pith.science/api/pith-number/MABHCAGCAX3Q5BM5IGFFGWJGU3/events.json","paper":"https://pith.science/paper/MABHCAGC"},"agent_actions":{"view_html":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3","download_json":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3.json","view_paper":"https://pith.science/paper/MABHCAGC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.11483&json=true","fetch_graph":"https://pith.science/api/pith-number/MABHCAGCAX3Q5BM5IGFFGWJGU3/graph.json","fetch_events":"https://pith.science/api/pith-number/MABHCAGCAX3Q5BM5IGFFGWJGU3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3/action/storage_attestation","attest_author":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3/action/author_attestation","sign_citation":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3/action/citation_signature","submit_replication":"https://pith.science/pith/MABHCAGCAX3Q5BM5IGFFGWJGU3/action/replication_record"}},"created_at":"2026-07-05T06:43:35.538663+00:00","updated_at":"2026-07-05T06:43:35.538663+00:00"}