{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DNGRGEBUZEM2M23A7UB7GBFIOD","short_pith_number":"pith:DNGRGEBU","schema_version":"1.0","canonical_sha256":"1b4d131034c919a66b60fd03f304a870eed6aa4863ad4c5f7d709e62f86d9f12","source":{"kind":"arxiv","id":"2407.09519","version":1},"attestation_state":"computed","paper":{"title":"Putting GPT-4o to the Sword: A Comprehensive Evaluation of Language, Vision, Speech, and Multimodal Proficiency","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Aashrith Mannuru, Brady Lund, Kadhim Hayawi, Laiba Batool, Muhammad Arbab Arshad, Nishith Reddy Mannuru, Ravi Varma Kumar Bevara, Sakib Shahriar","submitted_at":"2024-06-19T19:00:21Z","abstract_excerpt":"As large language models (LLMs) continue to advance, evaluating their comprehensive capabilities becomes significant for their application in various fields. This research study comprehensively evaluates the language, vision, speech, and multimodal capabilities of GPT-4o. The study employs standardized exam questions, reasoning tasks, and translation assessments to assess the model's language capability. Additionally, GPT-4o's vision and speech capabilities are tested through image classification and object recognition tasks, as well as accent classification. The multimodal evaluation assesses"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09519","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-19T19:00:21Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f13926244bd2b51c509b2bf4e644a3f7e49695bc9ee3efad7fbe7192a1e08875","abstract_canon_sha256":"11d7442d8e2a7666f96fd99820e51b7156a6d3dc1d4247eaa6ffd3041ed7e936"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:31.514801Z","signature_b64":"kSnnES6Qd5HJIOAqyMpKWz9WR+AAvpyFzI4EBESR/JGGVqY4s52INrrWSKv88b6OlKjartzyuA0P3JH2BdWaAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b4d131034c919a66b60fd03f304a870eed6aa4863ad4c5f7d709e62f86d9f12","last_reissued_at":"2026-07-05T08:43:31.514385Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:31.514385Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Putting GPT-4o to the Sword: A Comprehensive Evaluation of Language, Vision, Speech, and Multimodal Proficiency","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Aashrith Mannuru, Brady Lund, Kadhim Hayawi, Laiba Batool, Muhammad Arbab Arshad, Nishith Reddy Mannuru, Ravi Varma Kumar Bevara, Sakib Shahriar","submitted_at":"2024-06-19T19:00:21Z","abstract_excerpt":"As large language models (LLMs) continue to advance, evaluating their comprehensive capabilities becomes significant for their application in various fields. This research study comprehensively evaluates the language, vision, speech, and multimodal capabilities of GPT-4o. The study employs standardized exam questions, reasoning tasks, and translation assessments to assess the model's language capability. Additionally, GPT-4o's vision and speech capabilities are tested through image classification and object recognition tasks, as well as accent classification. The multimodal evaluation assesses"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09519","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09519/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09519","created_at":"2026-07-05T08:43:31.514443+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09519v1","created_at":"2026-07-05T08:43:31.514443+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09519","created_at":"2026-07-05T08:43:31.514443+00:00"},{"alias_kind":"pith_short_12","alias_value":"DNGRGEBUZEM2","created_at":"2026-07-05T08:43:31.514443+00:00"},{"alias_kind":"pith_short_16","alias_value":"DNGRGEBUZEM2M23A","created_at":"2026-07-05T08:43:31.514443+00:00"},{"alias_kind":"pith_short_8","alias_value":"DNGRGEBU","created_at":"2026-07-05T08:43:31.514443+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.05676","citing_title":"Nearly Solved? Robust Deepfake Detection Requires More than Visual Forensics","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD","json":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD.json","graph_json":"https://pith.science/api/pith-number/DNGRGEBUZEM2M23A7UB7GBFIOD/graph.json","events_json":"https://pith.science/api/pith-number/DNGRGEBUZEM2M23A7UB7GBFIOD/events.json","paper":"https://pith.science/paper/DNGRGEBU"},"agent_actions":{"view_html":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD","download_json":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD.json","view_paper":"https://pith.science/paper/DNGRGEBU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09519&json=true","fetch_graph":"https://pith.science/api/pith-number/DNGRGEBUZEM2M23A7UB7GBFIOD/graph.json","fetch_events":"https://pith.science/api/pith-number/DNGRGEBUZEM2M23A7UB7GBFIOD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD/action/storage_attestation","attest_author":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD/action/author_attestation","sign_citation":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD/action/citation_signature","submit_replication":"https://pith.science/pith/DNGRGEBUZEM2M23A7UB7GBFIOD/action/replication_record"}},"created_at":"2026-07-05T08:43:31.514443+00:00","updated_at":"2026-07-05T08:43:31.514443+00:00"}