{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CMOAKJFBQSVHSCKP27SNQORYJA","short_pith_number":"pith:CMOAKJFB","schema_version":"1.0","canonical_sha256":"131c0524a184aa79094fd7e4d83a384821c3f43e756ec74f7e9af81a945db65c","source":{"kind":"arxiv","id":"2303.08033","version":1},"attestation_state":"computed","paper":{"title":"Large Language Models (GPT) Struggle to Answer Multiple-Choice Questions about Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Arav Agarwal, Christopher Bogart, Jaromir Savelka, Majd Sakr","submitted_at":"2023-03-09T16:52:12Z","abstract_excerpt":"We analyzed effectiveness of three generative pre-trained transformer (GPT) models in answering multiple-choice question (MCQ) assessments, often involving short snippets of code, from introductory and intermediate programming courses at the postsecondary level. This emerging technology stirs countless discussions of its potential uses (e.g., exercise generation, code explanation) as well as misuses in programming education (e.g., cheating). However, the capabilities of GPT models and their limitations to reason about and/or analyze code in educational settings have been under-explored. We eva"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.08033","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-09T16:52:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d8ff0bd1ea1cd621701cbdca2da443b7105ec4bb5da98ae2c44fda93554f1693","abstract_canon_sha256":"779de9b430b9313ca20cd9ee1b0414a12899c239486c5b601e1836ef106599e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:14.244411Z","signature_b64":"e9ehUAYgMVEx9Q/VM6xavuawRy4kpqfa5FOgFaSm1YGhHzmMYpL2d0LmzwIP/lheJ4IxS8wyCijE2dMvwIZwCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"131c0524a184aa79094fd7e4d83a384821c3f43e756ec74f7e9af81a945db65c","last_reissued_at":"2026-07-05T05:51:14.243949Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:14.243949Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models (GPT) Struggle to Answer Multiple-Choice Questions about Code","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Arav Agarwal, Christopher Bogart, Jaromir Savelka, Majd Sakr","submitted_at":"2023-03-09T16:52:12Z","abstract_excerpt":"We analyzed effectiveness of three generative pre-trained transformer (GPT) models in answering multiple-choice question (MCQ) assessments, often involving short snippets of code, from introductory and intermediate programming courses at the postsecondary level. This emerging technology stirs countless discussions of its potential uses (e.g., exercise generation, code explanation) as well as misuses in programming education (e.g., cheating). However, the capabilities of GPT models and their limitations to reason about and/or analyze code in educational settings have been under-explored. We eva"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.08033","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.08033/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.08033","created_at":"2026-07-05T05:51:14.244014+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.08033v1","created_at":"2026-07-05T05:51:14.244014+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.08033","created_at":"2026-07-05T05:51:14.244014+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMOAKJFBQSVH","created_at":"2026-07-05T05:51:14.244014+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMOAKJFBQSVHSCKP","created_at":"2026-07-05T05:51:14.244014+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMOAKJFB","created_at":"2026-07-05T05:51:14.244014+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA","json":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA.json","graph_json":"https://pith.science/api/pith-number/CMOAKJFBQSVHSCKP27SNQORYJA/graph.json","events_json":"https://pith.science/api/pith-number/CMOAKJFBQSVHSCKP27SNQORYJA/events.json","paper":"https://pith.science/paper/CMOAKJFB"},"agent_actions":{"view_html":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA","download_json":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA.json","view_paper":"https://pith.science/paper/CMOAKJFB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.08033&json=true","fetch_graph":"https://pith.science/api/pith-number/CMOAKJFBQSVHSCKP27SNQORYJA/graph.json","fetch_events":"https://pith.science/api/pith-number/CMOAKJFBQSVHSCKP27SNQORYJA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA/action/storage_attestation","attest_author":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA/action/author_attestation","sign_citation":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA/action/citation_signature","submit_replication":"https://pith.science/pith/CMOAKJFBQSVHSCKP27SNQORYJA/action/replication_record"}},"created_at":"2026-07-05T05:51:14.244014+00:00","updated_at":"2026-07-05T05:51:14.244014+00:00"}