{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TCB3FNF6YP4JY64OIMOFO4NOH4","short_pith_number":"pith:TCB3FNF6","schema_version":"1.0","canonical_sha256":"9883b2b4bec3f89c7b8e431c5771ae3f24601d35611f95dd3abff8cd2c5d4b95","source":{"kind":"arxiv","id":"2310.13988","version":1},"attestation_state":"computed","paper":{"title":"GEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christian Federmann, Tom Kocmi","submitted_at":"2023-10-21T12:30:33Z","abstract_excerpt":"This paper introduces GEMBA-MQM, a GPT-based evaluation metric designed to detect translation quality errors, specifically for the quality estimation setting without the need for human reference translations. Based on the power of large language models (LLM), GEMBA-MQM employs a fixed three-shot prompting technique, querying the GPT-4 model to mark error quality spans. Compared to previous works, our method has language-agnostic prompts, thus avoiding the need for manual prompt preparation for new languages.\n  While preliminary results indicate that GEMBA-MQM achieves state-of-the-art accuracy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.13988","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-21T12:30:33Z","cross_cats_sorted":[],"title_canon_sha256":"89486b3a8803f8672c47b65ebaf7dbe578bf12d5a2f94a63216626d6f83ae1d7","abstract_canon_sha256":"4c02f9f6fef6482ea4c37b5d9d6ee8c3aca1a35400f4b4c5f4b3f7672e66de84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:37.919226Z","signature_b64":"Zm6wuRzqp3RQYxxk4rPiZvrCO2vBrOPKgP3AdlvW27x+JTUCSrLC0GGY8knB3SuuUtdDn+KqT+rMNclEyjwnAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9883b2b4bec3f89c7b8e431c5771ae3f24601d35611f95dd3abff8cd2c5d4b95","last_reissued_at":"2026-07-05T07:03:37.918818Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:37.918818Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christian Federmann, Tom Kocmi","submitted_at":"2023-10-21T12:30:33Z","abstract_excerpt":"This paper introduces GEMBA-MQM, a GPT-based evaluation metric designed to detect translation quality errors, specifically for the quality estimation setting without the need for human reference translations. Based on the power of large language models (LLM), GEMBA-MQM employs a fixed three-shot prompting technique, querying the GPT-4 model to mark error quality spans. Compared to previous works, our method has language-agnostic prompts, thus avoiding the need for manual prompt preparation for new languages.\n  While preliminary results indicate that GEMBA-MQM achieves state-of-the-art accuracy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.13988","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.13988/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.13988","created_at":"2026-07-05T07:03:37.918869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.13988v1","created_at":"2026-07-05T07:03:37.918869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.13988","created_at":"2026-07-05T07:03:37.918869+00:00"},{"alias_kind":"pith_short_12","alias_value":"TCB3FNF6YP4J","created_at":"2026-07-05T07:03:37.918869+00:00"},{"alias_kind":"pith_short_16","alias_value":"TCB3FNF6YP4JY64O","created_at":"2026-07-05T07:03:37.918869+00:00"},{"alias_kind":"pith_short_8","alias_value":"TCB3FNF6","created_at":"2026-07-05T07:03:37.918869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17041","citing_title":"Agentic AI Translate: An Agentic Translator Prototype for Translation as Communication Design","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17393","citing_title":"Who Watches the Watchmen? Humans Disagree With Translation Metrics on Unseen Domains","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4","json":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4.json","graph_json":"https://pith.science/api/pith-number/TCB3FNF6YP4JY64OIMOFO4NOH4/graph.json","events_json":"https://pith.science/api/pith-number/TCB3FNF6YP4JY64OIMOFO4NOH4/events.json","paper":"https://pith.science/paper/TCB3FNF6"},"agent_actions":{"view_html":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4","download_json":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4.json","view_paper":"https://pith.science/paper/TCB3FNF6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.13988&json=true","fetch_graph":"https://pith.science/api/pith-number/TCB3FNF6YP4JY64OIMOFO4NOH4/graph.json","fetch_events":"https://pith.science/api/pith-number/TCB3FNF6YP4JY64OIMOFO4NOH4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4/action/storage_attestation","attest_author":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4/action/author_attestation","sign_citation":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4/action/citation_signature","submit_replication":"https://pith.science/pith/TCB3FNF6YP4JY64OIMOFO4NOH4/action/replication_record"}},"created_at":"2026-07-05T07:03:37.918869+00:00","updated_at":"2026-07-05T07:03:37.918869+00:00"}