{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3QETYMSDU66JSAYUFENDZJCGUL","short_pith_number":"pith:3QETYMSD","schema_version":"1.0","canonical_sha256":"dc093c3243a7bc990314291a3ca446a2dbaa68f80f7b7afb9c056adc1412edf0","source":{"kind":"arxiv","id":"2309.04766","version":5},"attestation_state":"computed","paper":{"title":"SeaEval for Multilingual Foundation Models: From Cross-Lingual Alignment to Cultural Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aiti Aw, Bin Wang, Fangkai Jiao, Nancy F. Chen, Xin Huang, Yang Ding, Zhengyuan Liu","submitted_at":"2023-09-09T11:42:22Z","abstract_excerpt":"We present SeaEval, a benchmark for multilingual foundation models. In addition to characterizing how these models understand and reason with natural language, we also investigate how well they comprehend cultural practices, nuances, and values. Alongside standard accuracy metrics, we investigate the brittleness of foundation models in the dimensions of semantics and multilinguality. Our analyses span both open-sourced and closed models, leading to empirical results across classic NLP tasks, reasoning, and cultural comprehension. Key findings indicate (1) Most models exhibit varied behavior wh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.04766","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-09T11:42:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bc41b56a4f7f4862e48582f50606f0a5aea43bc2536b8be94c490bef39ef5ab7","abstract_canon_sha256":"daefae0e54479a1309cef4882994da49ac66700e68e6d8e0dad750db583fc2e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:30.195712Z","signature_b64":"vZGIvvRyLaGZfiaThzSw6fcNHwxGRcKkLZyq1QNXdKC6m79YIQTJSAO4XYpjRFglKKH2a4g0nJO9xDr80QZZCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc093c3243a7bc990314291a3ca446a2dbaa68f80f7b7afb9c056adc1412edf0","last_reissued_at":"2026-07-05T08:42:30.195221Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:30.195221Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SeaEval for Multilingual Foundation Models: From Cross-Lingual Alignment to Cultural Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aiti Aw, Bin Wang, Fangkai Jiao, Nancy F. Chen, Xin Huang, Yang Ding, Zhengyuan Liu","submitted_at":"2023-09-09T11:42:22Z","abstract_excerpt":"We present SeaEval, a benchmark for multilingual foundation models. In addition to characterizing how these models understand and reason with natural language, we also investigate how well they comprehend cultural practices, nuances, and values. Alongside standard accuracy metrics, we investigate the brittleness of foundation models in the dimensions of semantics and multilinguality. Our analyses span both open-sourced and closed models, leading to empirical results across classic NLP tasks, reasoning, and cultural comprehension. Key findings indicate (1) Most models exhibit varied behavior wh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.04766","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.04766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.04766","created_at":"2026-07-05T08:42:30.195276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.04766v5","created_at":"2026-07-05T08:42:30.195276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.04766","created_at":"2026-07-05T08:42:30.195276+00:00"},{"alias_kind":"pith_short_12","alias_value":"3QETYMSDU66J","created_at":"2026-07-05T08:42:30.195276+00:00"},{"alias_kind":"pith_short_16","alias_value":"3QETYMSDU66JSAYU","created_at":"2026-07-05T08:42:30.195276+00:00"},{"alias_kind":"pith_short_8","alias_value":"3QETYMSD","created_at":"2026-07-05T08:42:30.195276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05468","citing_title":"TASE: Token Awareness and Structured Evaluation for Multilingual Language Models","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL","json":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL.json","graph_json":"https://pith.science/api/pith-number/3QETYMSDU66JSAYUFENDZJCGUL/graph.json","events_json":"https://pith.science/api/pith-number/3QETYMSDU66JSAYUFENDZJCGUL/events.json","paper":"https://pith.science/paper/3QETYMSD"},"agent_actions":{"view_html":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL","download_json":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL.json","view_paper":"https://pith.science/paper/3QETYMSD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.04766&json=true","fetch_graph":"https://pith.science/api/pith-number/3QETYMSDU66JSAYUFENDZJCGUL/graph.json","fetch_events":"https://pith.science/api/pith-number/3QETYMSDU66JSAYUFENDZJCGUL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL/action/storage_attestation","attest_author":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL/action/author_attestation","sign_citation":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL/action/citation_signature","submit_replication":"https://pith.science/pith/3QETYMSDU66JSAYUFENDZJCGUL/action/replication_record"}},"created_at":"2026-07-05T08:42:30.195276+00:00","updated_at":"2026-07-05T08:42:30.195276+00:00"}