{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VRGBQRSWAD3WS6QOO2UZURZJY6","short_pith_number":"pith:VRGBQRSW","schema_version":"1.0","canonical_sha256":"ac4c18465600f7697a0e76a99a4729c7b1622eed4db77c5fd7eb703005b538cd","source":{"kind":"arxiv","id":"2410.10995","version":4},"attestation_state":"computed","paper":{"title":"Watching the Watchers: Exposing Gender Disparities in Machine Translation Quality Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andr\\'e F.T. Martins, Emmanouil Zaranis, Giuseppe Attanasio, Sweta Agrawal","submitted_at":"2024-10-14T18:24:52Z","abstract_excerpt":"Quality estimation (QE)-the automatic assessment of translation quality-has recently become crucial across several stages of the translation pipeline, from data curation to training and decoding. While QE metrics have been optimized to align with human judgments, whether they encode social biases has been largely overlooked. Biased QE risks favoring certain demographic groups over others, e.g., by exacerbating gaps in visibility and usability. This paper defines and investigates gender bias of QE metrics and discusses its downstream implications for machine translation (MT). Experiments with s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10995","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-14T18:24:52Z","cross_cats_sorted":[],"title_canon_sha256":"acceaac5d7890ff2677f230773bbf73988b97e81dee5b7c4b1623c573b45b5ab","abstract_canon_sha256":"d1041b184d5a8a57829ec61cae7987a12c35e528c38716438eb8d2df8545ca75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:25.058371Z","signature_b64":"GDz8Jx++iUQoDx9OCTSFSesTdcHoAsBzWjPLjd7qC7L8LxjR49Qi2SVzOaEgq80UYvWzxEaKI1gcs8mVkZC6DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac4c18465600f7697a0e76a99a4729c7b1622eed4db77c5fd7eb703005b538cd","last_reissued_at":"2026-07-05T11:14:25.057887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:25.057887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Watching the Watchers: Exposing Gender Disparities in Machine Translation Quality Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andr\\'e F.T. Martins, Emmanouil Zaranis, Giuseppe Attanasio, Sweta Agrawal","submitted_at":"2024-10-14T18:24:52Z","abstract_excerpt":"Quality estimation (QE)-the automatic assessment of translation quality-has recently become crucial across several stages of the translation pipeline, from data curation to training and decoding. While QE metrics have been optimized to align with human judgments, whether they encode social biases has been largely overlooked. Biased QE risks favoring certain demographic groups over others, e.g., by exacerbating gaps in visibility and usability. This paper defines and investigates gender bias of QE metrics and discusses its downstream implications for machine translation (MT). Experiments with s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10995","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10995/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10995","created_at":"2026-07-05T11:14:25.057946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10995v4","created_at":"2026-07-05T11:14:25.057946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10995","created_at":"2026-07-05T11:14:25.057946+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRGBQRSWAD3W","created_at":"2026-07-05T11:14:25.057946+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRGBQRSWAD3WS6QO","created_at":"2026-07-05T11:14:25.057946+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRGBQRSW","created_at":"2026-07-05T11:14:25.057946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.11615","citing_title":"MT-LENS: An all-in-one Toolkit for Better Machine Translation Evaluation","ref_index":47,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6","json":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6.json","graph_json":"https://pith.science/api/pith-number/VRGBQRSWAD3WS6QOO2UZURZJY6/graph.json","events_json":"https://pith.science/api/pith-number/VRGBQRSWAD3WS6QOO2UZURZJY6/events.json","paper":"https://pith.science/paper/VRGBQRSW"},"agent_actions":{"view_html":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6","download_json":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6.json","view_paper":"https://pith.science/paper/VRGBQRSW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10995&json=true","fetch_graph":"https://pith.science/api/pith-number/VRGBQRSWAD3WS6QOO2UZURZJY6/graph.json","fetch_events":"https://pith.science/api/pith-number/VRGBQRSWAD3WS6QOO2UZURZJY6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6/action/storage_attestation","attest_author":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6/action/author_attestation","sign_citation":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6/action/citation_signature","submit_replication":"https://pith.science/pith/VRGBQRSWAD3WS6QOO2UZURZJY6/action/replication_record"}},"created_at":"2026-07-05T11:14:25.057946+00:00","updated_at":"2026-07-05T11:14:25.057946+00:00"}