{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FIEY5VINW5OIZXZ4JVUL3L5GLS","short_pith_number":"pith:FIEY5VIN","schema_version":"1.0","canonical_sha256":"2a098ed50db75c8cdf3c4d68bdafa65c9769990336cf66b55e60b7bb85fc24aa","source":{"kind":"arxiv","id":"2402.05374","version":5},"attestation_state":"computed","paper":{"title":"CIC: A Framework for Culturally-Aware Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jihie Kim, Youngsik Yun","submitted_at":"2024-02-08T03:12:25Z","abstract_excerpt":"Image Captioning generates descriptive sentences from images using Vision-Language Pre-trained models (VLPs) such as BLIP, which has improved greatly. However, current methods lack the generation of detailed descriptive captions for the cultural elements depicted in the images, such as the traditional clothing worn by people from Asian cultural groups. In this paper, we propose a new framework, Culturally-aware Image Captioning (CIC), that generates captions and describes cultural elements extracted from cultural visual elements in images representing cultures. Inspired by methods combining vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05374","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-08T03:12:25Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"bf433bf5d65b3035e451a5ee43ec233faad0693093e9c0d34ab10620527ee0c6","abstract_canon_sha256":"f33e013995ae46c9c0f1c1a4c4e131608cb62bfd6e62d81e1541a9eac19287a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:34.869780Z","signature_b64":"ipCp/D/DarNQ/2zSEXC4Z6eSeZh33ORBiGtbNJ65qx16ShOeD0XZn2l99v0Kw4zz8iSod2pdm02yxVrdLzP1Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a098ed50db75c8cdf3c4d68bdafa65c9769990336cf66b55e60b7bb85fc24aa","last_reissued_at":"2026-07-05T10:18:34.869303Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:34.869303Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CIC: A Framework for Culturally-Aware Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jihie Kim, Youngsik Yun","submitted_at":"2024-02-08T03:12:25Z","abstract_excerpt":"Image Captioning generates descriptive sentences from images using Vision-Language Pre-trained models (VLPs) such as BLIP, which has improved greatly. However, current methods lack the generation of detailed descriptive captions for the cultural elements depicted in the images, such as the traditional clothing worn by people from Asian cultural groups. In this paper, we propose a new framework, Culturally-aware Image Captioning (CIC), that generates captions and describes cultural elements extracted from cultural visual elements in images representing cultures. Inspired by methods combining vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05374","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05374/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05374","created_at":"2026-07-05T10:18:34.869370+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05374v5","created_at":"2026-07-05T10:18:34.869370+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05374","created_at":"2026-07-05T10:18:34.869370+00:00"},{"alias_kind":"pith_short_12","alias_value":"FIEY5VINW5OI","created_at":"2026-07-05T10:18:34.869370+00:00"},{"alias_kind":"pith_short_16","alias_value":"FIEY5VINW5OIZXZ4","created_at":"2026-07-05T10:18:34.869370+00:00"},{"alias_kind":"pith_short_8","alias_value":"FIEY5VIN","created_at":"2026-07-05T10:18:34.869370+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30800","citing_title":"Computer-Aided Tagging on Wikimedia Commons: Designing for Human-AI Collaboration in Open Knowledge Work","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS","json":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS.json","graph_json":"https://pith.science/api/pith-number/FIEY5VINW5OIZXZ4JVUL3L5GLS/graph.json","events_json":"https://pith.science/api/pith-number/FIEY5VINW5OIZXZ4JVUL3L5GLS/events.json","paper":"https://pith.science/paper/FIEY5VIN"},"agent_actions":{"view_html":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS","download_json":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS.json","view_paper":"https://pith.science/paper/FIEY5VIN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05374&json=true","fetch_graph":"https://pith.science/api/pith-number/FIEY5VINW5OIZXZ4JVUL3L5GLS/graph.json","fetch_events":"https://pith.science/api/pith-number/FIEY5VINW5OIZXZ4JVUL3L5GLS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS/action/storage_attestation","attest_author":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS/action/author_attestation","sign_citation":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS/action/citation_signature","submit_replication":"https://pith.science/pith/FIEY5VINW5OIZXZ4JVUL3L5GLS/action/replication_record"}},"created_at":"2026-07-05T10:18:34.869370+00:00","updated_at":"2026-07-05T10:18:34.869370+00:00"}