{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:O3JLXYY2BPQKR4KCCNJVSFWSQT","short_pith_number":"pith:O3JLXYY2","schema_version":"1.0","canonical_sha256":"76d2bbe31a0be0a8f14213535916d284f36bb6643014f7969fc38110abf51e79","source":{"kind":"arxiv","id":"2305.14711","version":3},"attestation_state":"computed","paper":{"title":"Gender Biases in Automatic Evaluation Metrics for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Asli Celikyilmaz, Haoyi Qiu, Nanyun Peng, Tianlu Wang, Zi-Yi Dou","submitted_at":"2023-05-24T04:27:40Z","abstract_excerpt":"Model-based evaluation metrics (e.g., CLIPScore and GPTScore) have demonstrated decent correlations with human judgments in various language generation tasks. However, their impact on fairness remains largely unexplored. It is widely recognized that pretrained models can inadvertently encode societal biases, thus employing these models for evaluation purposes may inadvertently perpetuate and amplify biases. For example, an evaluation metric may favor the caption \"a woman is calculating an account book\" over \"a man is calculating an account book,\" even if the image only shows male accountants. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14711","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T04:27:40Z","cross_cats_sorted":[],"title_canon_sha256":"343a271689bdae1e3fe866b55f1115ed972069dbf468b5764d4e9263ee84dd97","abstract_canon_sha256":"bbb3069a5796b458b74f4c4673ac4ca0b09ed8266ae1a3dbabe7365dd2890da3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:28.529188Z","signature_b64":"Gy+c+YECLjzjyDQb9EYNbzgd1jxEdu7EL+bf6A4VEtmt/IdErPVnX+mbCD3JYYJLNUKdW5LQk9BAvlW1z8R4AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76d2bbe31a0be0a8f14213535916d284f36bb6643014f7969fc38110abf51e79","last_reissued_at":"2026-07-05T07:08:28.528713Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:28.528713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Gender Biases in Automatic Evaluation Metrics for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Asli Celikyilmaz, Haoyi Qiu, Nanyun Peng, Tianlu Wang, Zi-Yi Dou","submitted_at":"2023-05-24T04:27:40Z","abstract_excerpt":"Model-based evaluation metrics (e.g., CLIPScore and GPTScore) have demonstrated decent correlations with human judgments in various language generation tasks. However, their impact on fairness remains largely unexplored. It is widely recognized that pretrained models can inadvertently encode societal biases, thus employing these models for evaluation purposes may inadvertently perpetuate and amplify biases. For example, an evaluation metric may favor the caption \"a woman is calculating an account book\" over \"a man is calculating an account book,\" even if the image only shows male accountants. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14711","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14711/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14711","created_at":"2026-07-05T07:08:28.528770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14711v3","created_at":"2026-07-05T07:08:28.528770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14711","created_at":"2026-07-05T07:08:28.528770+00:00"},{"alias_kind":"pith_short_12","alias_value":"O3JLXYY2BPQK","created_at":"2026-07-05T07:08:28.528770+00:00"},{"alias_kind":"pith_short_16","alias_value":"O3JLXYY2BPQKR4KC","created_at":"2026-07-05T07:08:28.528770+00:00"},{"alias_kind":"pith_short_8","alias_value":"O3JLXYY2","created_at":"2026-07-05T07:08:28.528770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.07878","citing_title":"A Woman with a Knife or A Knife with a Woman? Measuring Directional Bias Amplification in Image Captions","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT","json":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT.json","graph_json":"https://pith.science/api/pith-number/O3JLXYY2BPQKR4KCCNJVSFWSQT/graph.json","events_json":"https://pith.science/api/pith-number/O3JLXYY2BPQKR4KCCNJVSFWSQT/events.json","paper":"https://pith.science/paper/O3JLXYY2"},"agent_actions":{"view_html":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT","download_json":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT.json","view_paper":"https://pith.science/paper/O3JLXYY2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14711&json=true","fetch_graph":"https://pith.science/api/pith-number/O3JLXYY2BPQKR4KCCNJVSFWSQT/graph.json","fetch_events":"https://pith.science/api/pith-number/O3JLXYY2BPQKR4KCCNJVSFWSQT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT/action/storage_attestation","attest_author":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT/action/author_attestation","sign_citation":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT/action/citation_signature","submit_replication":"https://pith.science/pith/O3JLXYY2BPQKR4KCCNJVSFWSQT/action/replication_record"}},"created_at":"2026-07-05T07:08:28.528770+00:00","updated_at":"2026-07-05T07:08:28.528770+00:00"}