{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q7MT3W56HGCJVGSOIIF67WVSNT","short_pith_number":"pith:Q7MT3W56","schema_version":"1.0","canonical_sha256":"87d93ddbbe39849a9a4e420befdab26cc77f41f79d6b063e9cc45bc555657855","source":{"kind":"arxiv","id":"2502.17797","version":1},"attestation_state":"computed","paper":{"title":"Enhancing Human Evaluation in Machine Translation with Comparative Judgment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daniel Deutsch, Markus Freitag, Parker Riley, Yixiao Song","submitted_at":"2025-02-25T03:02:24Z","abstract_excerpt":"Human evaluation is crucial for assessing rapidly evolving language models but is influenced by annotator proficiency and task design. This study explores the integration of comparative judgment into human annotation for machine translation (MT) and evaluates three annotation setups-point-wise Multidimensional Quality Metrics (MQM), side-by-side (SxS) MQM, and its simplified version SxS relative ranking (RR). In MQM, annotators mark error spans with categories and severity levels. SxS MQM extends MQM to pairwise error annotation for two translations of the same input, while SxS RR focuses on s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17797","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-25T03:02:24Z","cross_cats_sorted":[],"title_canon_sha256":"48950608d27703556f9d044c5e1742df520b5f0e804cb6aa4f4ee7fb2203312d","abstract_canon_sha256":"9b074eaf131aff56c17ac29c8c82682ffb0ba4b0e9f6e91c119a65bc92df8b64"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:44.249730Z","signature_b64":"RkI8/56UXagY4PvEwaaXw3LYFYWOIuqmLMzAH7EVIuAg3TiRCO0ZxyVAVEjMrQdxfWDKMbKd3ybJ0JCVrKmpBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87d93ddbbe39849a9a4e420befdab26cc77f41f79d6b063e9cc45bc555657855","last_reissued_at":"2026-07-05T10:19:44.249230Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:44.249230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Human Evaluation in Machine Translation with Comparative Judgment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daniel Deutsch, Markus Freitag, Parker Riley, Yixiao Song","submitted_at":"2025-02-25T03:02:24Z","abstract_excerpt":"Human evaluation is crucial for assessing rapidly evolving language models but is influenced by annotator proficiency and task design. This study explores the integration of comparative judgment into human annotation for machine translation (MT) and evaluates three annotation setups-point-wise Multidimensional Quality Metrics (MQM), side-by-side (SxS) MQM, and its simplified version SxS relative ranking (RR). In MQM, annotators mark error spans with categories and severity levels. SxS MQM extends MQM to pairwise error annotation for two translations of the same input, while SxS RR focuses on s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17797","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17797/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17797","created_at":"2026-07-05T10:19:44.249296+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17797v1","created_at":"2026-07-05T10:19:44.249296+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17797","created_at":"2026-07-05T10:19:44.249296+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q7MT3W56HGCJ","created_at":"2026-07-05T10:19:44.249296+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q7MT3W56HGCJVGSO","created_at":"2026-07-05T10:19:44.249296+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q7MT3W56","created_at":"2026-07-05T10:19:44.249296+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.16378","citing_title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT","json":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT.json","graph_json":"https://pith.science/api/pith-number/Q7MT3W56HGCJVGSOIIF67WVSNT/graph.json","events_json":"https://pith.science/api/pith-number/Q7MT3W56HGCJVGSOIIF67WVSNT/events.json","paper":"https://pith.science/paper/Q7MT3W56"},"agent_actions":{"view_html":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT","download_json":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT.json","view_paper":"https://pith.science/paper/Q7MT3W56","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17797&json=true","fetch_graph":"https://pith.science/api/pith-number/Q7MT3W56HGCJVGSOIIF67WVSNT/graph.json","fetch_events":"https://pith.science/api/pith-number/Q7MT3W56HGCJVGSOIIF67WVSNT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT/action/storage_attestation","attest_author":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT/action/author_attestation","sign_citation":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT/action/citation_signature","submit_replication":"https://pith.science/pith/Q7MT3W56HGCJVGSOIIF67WVSNT/action/replication_record"}},"created_at":"2026-07-05T10:19:44.249296+00:00","updated_at":"2026-07-05T10:19:44.249296+00:00"}