{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E3FFWJYDIUO35PPEIYMVEKZ2OS","short_pith_number":"pith:E3FFWJYD","schema_version":"1.0","canonical_sha256":"26ca5b2703451dbebde44619522b3a748e2a5a82fba5fd73360f642cf7eebf40","source":{"kind":"arxiv","id":"2505.22793","version":2},"attestation_state":"computed","paper":{"title":"Evaluation of Cultural Competence of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Antonia Karamolegkou, Daniel Hershcovich, Ekaterina Shutova, Jiaang Li, Lauren Tilton, Maria Antoniak, Negar Rostamzadeh, Serge Belongie, Siddhesh Milind Pawar, Srishti Yadav, Stella Frank, Taylor Arnold, Zhaochong An","submitted_at":"2025-05-28T19:04:04Z","abstract_excerpt":"Modern vision-language models (VLMs) often fail at cultural competency evaluations and benchmarks. Given the diversity of applications built upon VLMs, there is renewed interest in understanding how they encode cultural nuances. While individual aspects of this problem have been studied, we still lack a comprehensive framework for systematically identifying and annotating the nuanced cultural dimensions present in images for VLMs. This position paper argues that foundational methodologies from visual culture studies (cultural studies, semiotics, and visual studies) are necessary for cultural a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22793","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-28T19:04:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"9cbe59ef4cbd1765cc37b3e496e6706a8f2a87634aefec07abd34b53c5f8dab5","abstract_canon_sha256":"5258141c3c597d2fd27a4e8b72f7c889cfdeb7d70a87ac7e499371ec81ffca5d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:47.260473Z","signature_b64":"hqUZVUjt0uPYO4lZ820lUj2q/AW13+ZSW1LOfCpuV7JeVoNguUy5qefpRIlJe4UQyPVQEP7vA3HKrhXOZ/bLDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26ca5b2703451dbebde44619522b3a748e2a5a82fba5fd73360f642cf7eebf40","last_reissued_at":"2026-07-05T11:53:47.259991Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:47.259991Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of Cultural Competence of Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Antonia Karamolegkou, Daniel Hershcovich, Ekaterina Shutova, Jiaang Li, Lauren Tilton, Maria Antoniak, Negar Rostamzadeh, Serge Belongie, Siddhesh Milind Pawar, Srishti Yadav, Stella Frank, Taylor Arnold, Zhaochong An","submitted_at":"2025-05-28T19:04:04Z","abstract_excerpt":"Modern vision-language models (VLMs) often fail at cultural competency evaluations and benchmarks. Given the diversity of applications built upon VLMs, there is renewed interest in understanding how they encode cultural nuances. While individual aspects of this problem have been studied, we still lack a comprehensive framework for systematically identifying and annotating the nuanced cultural dimensions present in images for VLMs. This position paper argues that foundational methodologies from visual culture studies (cultural studies, semiotics, and visual studies) are necessary for cultural a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22793","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22793","created_at":"2026-07-05T11:53:47.260048+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22793v2","created_at":"2026-07-05T11:53:47.260048+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22793","created_at":"2026-07-05T11:53:47.260048+00:00"},{"alias_kind":"pith_short_12","alias_value":"E3FFWJYDIUO3","created_at":"2026-07-05T11:53:47.260048+00:00"},{"alias_kind":"pith_short_16","alias_value":"E3FFWJYDIUO35PPE","created_at":"2026-07-05T11:53:47.260048+00:00"},{"alias_kind":"pith_short_8","alias_value":"E3FFWJYD","created_at":"2026-07-05T11:53:47.260048+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.07763","citing_title":"Jako Tako or Fluent? Presenting PoVisLE: A Polish Vision-Language Evaluation","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS","json":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS.json","graph_json":"https://pith.science/api/pith-number/E3FFWJYDIUO35PPEIYMVEKZ2OS/graph.json","events_json":"https://pith.science/api/pith-number/E3FFWJYDIUO35PPEIYMVEKZ2OS/events.json","paper":"https://pith.science/paper/E3FFWJYD"},"agent_actions":{"view_html":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS","download_json":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS.json","view_paper":"https://pith.science/paper/E3FFWJYD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22793&json=true","fetch_graph":"https://pith.science/api/pith-number/E3FFWJYDIUO35PPEIYMVEKZ2OS/graph.json","fetch_events":"https://pith.science/api/pith-number/E3FFWJYDIUO35PPEIYMVEKZ2OS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS/action/storage_attestation","attest_author":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS/action/author_attestation","sign_citation":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS/action/citation_signature","submit_replication":"https://pith.science/pith/E3FFWJYDIUO35PPEIYMVEKZ2OS/action/replication_record"}},"created_at":"2026-07-05T11:53:47.260048+00:00","updated_at":"2026-07-05T11:53:47.260048+00:00"}