{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AGFUTZ7S77PWZNJPHGZTHZ2R7Z","short_pith_number":"pith:AGFUTZ7S","schema_version":"1.0","canonical_sha256":"018b49e7f2ffdf6cb52f39b333e751fe7aa515347e5b5d1d99b1329fb308f205","source":{"kind":"arxiv","id":"2401.02985","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Large Language Models on the GMAT: Implications for the Future of Business Education","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jordan W. Suchow, Necdet G\\\"urkan, Vahid Ashrafimoghari","submitted_at":"2024-01-02T03:54:50Z","abstract_excerpt":"The rapid evolution of artificial intelligence (AI), especially in the domain of Large Language Models (LLMs) and generative AI, has opened new avenues for application across various fields, yet its role in business education remains underexplored. This study introduces the first benchmark to assess the performance of seven major LLMs, OpenAI's models (GPT-3.5 Turbo, GPT-4, and GPT-4 Turbo), Google's models (PaLM 2, Gemini 1.0 Pro), and Anthropic's models (Claude 2 and Claude 2.1), on the GMAT, which is a key exam in the admission process for graduate business programs. Our analysis shows that"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.02985","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-02T03:54:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"981fa40f4907c752c6d8cd22c8d4bc5416f9718db4157a1f73ddd47dbfbb00b6","abstract_canon_sha256":"c76c48b544237b59fb22eace95041430184e79ce86e7bc2018710c3142ba0364"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:31:00.464901Z","signature_b64":"2K4yRim5YXefCoIjiDRrRJ7e4bY8/RtYjTgD1FcnSBc+Hlz1xuooz7kRhczaol9dBuIbcIQFk4t6rMmbstXpDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"018b49e7f2ffdf6cb52f39b333e751fe7aa515347e5b5d1d99b1329fb308f205","last_reissued_at":"2026-07-05T07:31:00.464576Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:31:00.464576Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Large Language Models on the GMAT: Implications for the Future of Business Education","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jordan W. Suchow, Necdet G\\\"urkan, Vahid Ashrafimoghari","submitted_at":"2024-01-02T03:54:50Z","abstract_excerpt":"The rapid evolution of artificial intelligence (AI), especially in the domain of Large Language Models (LLMs) and generative AI, has opened new avenues for application across various fields, yet its role in business education remains underexplored. This study introduces the first benchmark to assess the performance of seven major LLMs, OpenAI's models (GPT-3.5 Turbo, GPT-4, and GPT-4 Turbo), Google's models (PaLM 2, Gemini 1.0 Pro), and Anthropic's models (Claude 2 and Claude 2.1), on the GMAT, which is a key exam in the admission process for graduate business programs. Our analysis shows that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.02985","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.02985","created_at":"2026-07-05T07:31:00.464628+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.02985v1","created_at":"2026-07-05T07:31:00.464628+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02985","created_at":"2026-07-05T07:31:00.464628+00:00"},{"alias_kind":"pith_short_12","alias_value":"AGFUTZ7S77PW","created_at":"2026-07-05T07:31:00.464628+00:00"},{"alias_kind":"pith_short_16","alias_value":"AGFUTZ7S77PWZNJP","created_at":"2026-07-05T07:31:00.464628+00:00"},{"alias_kind":"pith_short_8","alias_value":"AGFUTZ7S","created_at":"2026-07-05T07:31:00.464628+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00049","citing_title":"Prompting GPT-5 on Scrum Certification Questions: An Empirical Accuracy Study","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00048","citing_title":"Comparing Large Language Models on Scrum Certification-Style Questions: Accuracy, Stability, and Error Patterns","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z","json":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z.json","graph_json":"https://pith.science/api/pith-number/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/graph.json","events_json":"https://pith.science/api/pith-number/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/events.json","paper":"https://pith.science/paper/AGFUTZ7S"},"agent_actions":{"view_html":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z","download_json":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z.json","view_paper":"https://pith.science/paper/AGFUTZ7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.02985&json=true","fetch_graph":"https://pith.science/api/pith-number/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/action/storage_attestation","attest_author":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/action/author_attestation","sign_citation":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/action/citation_signature","submit_replication":"https://pith.science/pith/AGFUTZ7S77PWZNJPHGZTHZ2R7Z/action/replication_record"}},"created_at":"2026-07-05T07:31:00.464628+00:00","updated_at":"2026-07-05T07:31:00.464628+00:00"}