{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K7URBVKNFDKEGPO6IQLINQBD2U","short_pith_number":"pith:K7URBVKN","schema_version":"1.0","canonical_sha256":"57e910d54d28d4433dde441686c023d53178f73b1d56b6004c88f9f61cac8aed","source":{"kind":"arxiv","id":"2407.00263","version":1},"attestation_state":"computed","paper":{"title":"From Local Concepts to Universals: Evaluating the Multicultural Understanding of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Aditya Chinchure, EunJeong Hwang, Mehar Bhatia, Sahithya Ravi, Vered Shwartz","submitted_at":"2024-06-28T23:28:28Z","abstract_excerpt":"Despite recent advancements in vision-language models, their performance remains suboptimal on images from non-western cultures due to underrepresentation in training datasets. Various benchmarks have been proposed to test models' cultural inclusivity, but they have limited coverage of cultures and do not adequately assess cultural diversity across universal as well as culture-specific local concepts. To address these limitations, we introduce the GlobalRG benchmark, comprising two challenging tasks: retrieval across universals and cultural visual grounding. The former task entails retrieving "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00263","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-28T23:28:28Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"44f7cbaea7d3165a1b62c5339ac66cdcc762b5da0882dd861f658e6cba2ae520","abstract_canon_sha256":"3e9d5b04676ea2f56f020f7c1964590097882798c46797860c1098cf7de5d374"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:16.861658Z","signature_b64":"ghfULyQxD7SXM2MZwonnOF5oEFiELVd5kz49sDrCcIJCIdd285XL16vzOpOBeiBx/10+Th1qSR8H4i+lahhMCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57e910d54d28d4433dde441686c023d53178f73b1d56b6004c88f9f61cac8aed","last_reissued_at":"2026-07-05T08:38:16.861241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:16.861241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Local Concepts to Universals: Evaluating the Multicultural Understanding of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Aditya Chinchure, EunJeong Hwang, Mehar Bhatia, Sahithya Ravi, Vered Shwartz","submitted_at":"2024-06-28T23:28:28Z","abstract_excerpt":"Despite recent advancements in vision-language models, their performance remains suboptimal on images from non-western cultures due to underrepresentation in training datasets. Various benchmarks have been proposed to test models' cultural inclusivity, but they have limited coverage of cultures and do not adequately assess cultural diversity across universal as well as culture-specific local concepts. To address these limitations, we introduce the GlobalRG benchmark, comprising two challenging tasks: retrieval across universals and cultural visual grounding. The former task entails retrieving "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00263","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00263/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00263","created_at":"2026-07-05T08:38:16.861295+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00263v1","created_at":"2026-07-05T08:38:16.861295+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00263","created_at":"2026-07-05T08:38:16.861295+00:00"},{"alias_kind":"pith_short_12","alias_value":"K7URBVKNFDKE","created_at":"2026-07-05T08:38:16.861295+00:00"},{"alias_kind":"pith_short_16","alias_value":"K7URBVKNFDKEGPO6","created_at":"2026-07-05T08:38:16.861295+00:00"},{"alias_kind":"pith_short_8","alias_value":"K7URBVKN","created_at":"2026-07-05T08:38:16.861295+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03345","citing_title":"Beyond Semantics: Modeling Factual and Affective Perceptual Experiences from Vision-Language Data","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U","json":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U.json","graph_json":"https://pith.science/api/pith-number/K7URBVKNFDKEGPO6IQLINQBD2U/graph.json","events_json":"https://pith.science/api/pith-number/K7URBVKNFDKEGPO6IQLINQBD2U/events.json","paper":"https://pith.science/paper/K7URBVKN"},"agent_actions":{"view_html":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U","download_json":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U.json","view_paper":"https://pith.science/paper/K7URBVKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00263&json=true","fetch_graph":"https://pith.science/api/pith-number/K7URBVKNFDKEGPO6IQLINQBD2U/graph.json","fetch_events":"https://pith.science/api/pith-number/K7URBVKNFDKEGPO6IQLINQBD2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U/action/storage_attestation","attest_author":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U/action/author_attestation","sign_citation":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U/action/citation_signature","submit_replication":"https://pith.science/pith/K7URBVKNFDKEGPO6IQLINQBD2U/action/replication_record"}},"created_at":"2026-07-05T08:38:16.861295+00:00","updated_at":"2026-07-05T08:38:16.861295+00:00"}