{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:54NMQOJWG6KSOZ5NUDZTYSIZL5","short_pith_number":"pith:54NMQOJW","schema_version":"1.0","canonical_sha256":"ef1ac8393637952767ada0f33c49195f40f559e729bc996a801f8f2ab93510e8","source":{"kind":"arxiv","id":"2303.12528","version":4},"attestation_state":"computed","paper":{"title":"MEGA: Multilingual Evaluation of Generative AI","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akshay Nambi, Harshita Diddee, Kabir Ahuja, Kalika Bali, Krithika Ramesh, Maxamed Axmed, Millicent Ochieng, Prachi Jain, Rishav Hada, Sameer Segal, Sunayana Sitaram, Tanuja Ganu","submitted_at":"2023-03-22T13:03:10Z","abstract_excerpt":"Generative AI models have shown impressive performance on many Natural Language Processing tasks such as language understanding, reasoning, and language generation. An important question being asked by the AI community today is about the capabilities and limits of these models, and it is clear that evaluating generative AI is very challenging. Most studies on generative LLMs have been restricted to English and it is unclear how capable these models are at understanding and generating text in other languages. We present the first comprehensive benchmarking of generative LLMs - MEGA, which evalu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.12528","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-22T13:03:10Z","cross_cats_sorted":[],"title_canon_sha256":"dc8915497484a29004af4388de270a1b938335a28df3549dc341afefd822dc4b","abstract_canon_sha256":"5480f0ca15631e992d0bfc16f3631581bd309d8700978568d13e35aa2e680d8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:22.657448Z","signature_b64":"M/10InFUYMD/83rrRZ21ydZaiJ4K2jVhbokpo/HQm3Qv+K/oEBWfwrzzGrxzYOxKA8dQd54+gS8nnqbGp/yrBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef1ac8393637952767ada0f33c49195f40f559e729bc996a801f8f2ab93510e8","last_reissued_at":"2026-07-05T07:03:22.656958Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:22.656958Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MEGA: Multilingual Evaluation of Generative AI","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akshay Nambi, Harshita Diddee, Kabir Ahuja, Kalika Bali, Krithika Ramesh, Maxamed Axmed, Millicent Ochieng, Prachi Jain, Rishav Hada, Sameer Segal, Sunayana Sitaram, Tanuja Ganu","submitted_at":"2023-03-22T13:03:10Z","abstract_excerpt":"Generative AI models have shown impressive performance on many Natural Language Processing tasks such as language understanding, reasoning, and language generation. An important question being asked by the AI community today is about the capabilities and limits of these models, and it is clear that evaluating generative AI is very challenging. Most studies on generative LLMs have been restricted to English and it is unclear how capable these models are at understanding and generating text in other languages. We present the first comprehensive benchmarking of generative LLMs - MEGA, which evalu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.12528","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.12528/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.12528","created_at":"2026-07-05T07:03:22.657015+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.12528v4","created_at":"2026-07-05T07:03:22.657015+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.12528","created_at":"2026-07-05T07:03:22.657015+00:00"},{"alias_kind":"pith_short_12","alias_value":"54NMQOJWG6KS","created_at":"2026-07-05T07:03:22.657015+00:00"},{"alias_kind":"pith_short_16","alias_value":"54NMQOJWG6KSOZ5N","created_at":"2026-07-05T07:03:22.657015+00:00"},{"alias_kind":"pith_short_8","alias_value":"54NMQOJW","created_at":"2026-07-05T07:03:22.657015+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21645","citing_title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","ref_index":247,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10657","citing_title":"Are We Evaluating Knowledge or Phrasing? Mitigating MCQA Sensitivity with ParaEval","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25141","citing_title":"LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14774","citing_title":"LiveCLKTBench: Towards Reliable Evaluation of Cross-Lingual Knowledge Transfer in Multilingual LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":214,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5","json":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5.json","graph_json":"https://pith.science/api/pith-number/54NMQOJWG6KSOZ5NUDZTYSIZL5/graph.json","events_json":"https://pith.science/api/pith-number/54NMQOJWG6KSOZ5NUDZTYSIZL5/events.json","paper":"https://pith.science/paper/54NMQOJW"},"agent_actions":{"view_html":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5","download_json":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5.json","view_paper":"https://pith.science/paper/54NMQOJW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.12528&json=true","fetch_graph":"https://pith.science/api/pith-number/54NMQOJWG6KSOZ5NUDZTYSIZL5/graph.json","fetch_events":"https://pith.science/api/pith-number/54NMQOJWG6KSOZ5NUDZTYSIZL5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5/action/storage_attestation","attest_author":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5/action/author_attestation","sign_citation":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5/action/citation_signature","submit_replication":"https://pith.science/pith/54NMQOJWG6KSOZ5NUDZTYSIZL5/action/replication_record"}},"created_at":"2026-07-05T07:03:22.657015+00:00","updated_at":"2026-07-05T07:03:22.657015+00:00"}