{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5L5TZGO4U4QOX4RVOLKQ3WE5PF","short_pith_number":"pith:5L5TZGO4","schema_version":"1.0","canonical_sha256":"eafb3c99dca720ebf23572d50dd89d796f26778fea72ea9088c49e64c7b28709","source":{"kind":"arxiv","id":"2409.14335","version":2},"attestation_state":"computed","paper":{"title":"MQM-APE: Toward High-Quality Error Annotation Predictors with Automatic Post-Editing in LLM Translation Evaluators","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Jinxia Zhang, Kanjian Zhang, Liang Ding, Qingyu Lu","submitted_at":"2024-09-22T06:43:40Z","abstract_excerpt":"Large Language Models (LLMs) have shown significant potential as judges for Machine Translation (MT) quality assessment, providing both scores and fine-grained feedback. Although approaches such as GEMBA-MQM have shown state-of-the-art performance on reference-free evaluation, the predicted errors do not align well with those annotated by human, limiting their interpretability as feedback signals. To enhance the quality of error annotations predicted by LLM evaluators, we introduce a universal and training-free framework, $\\textbf{MQM-APE}$, based on the idea of filtering out non-impactful err"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14335","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-22T06:43:40Z","cross_cats_sorted":[],"title_canon_sha256":"edf39792d49567ef1fdd1aa041947b4863bbcad42808e4e4eb92ed8402fc105d","abstract_canon_sha256":"d2bb5a49101587b9ed021aed57f97c67d286eebcfaf162ae62f272ac144b6569"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:10.453621Z","signature_b64":"Bu6v0PjLICVtAR6vhIoud4QotaQchNvqBgjuIVm/F6lV5Kb7inuY5NDZvi1xHq+6JFGPmhtB/XFp4KyxAczuDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eafb3c99dca720ebf23572d50dd89d796f26778fea72ea9088c49e64c7b28709","last_reissued_at":"2026-07-05T09:49:10.453127Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:10.453127Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MQM-APE: Toward High-Quality Error Annotation Predictors with Automatic Post-Editing in LLM Translation Evaluators","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Jinxia Zhang, Kanjian Zhang, Liang Ding, Qingyu Lu","submitted_at":"2024-09-22T06:43:40Z","abstract_excerpt":"Large Language Models (LLMs) have shown significant potential as judges for Machine Translation (MT) quality assessment, providing both scores and fine-grained feedback. Although approaches such as GEMBA-MQM have shown state-of-the-art performance on reference-free evaluation, the predicted errors do not align well with those annotated by human, limiting their interpretability as feedback signals. To enhance the quality of error annotations predicted by LLM evaluators, we introduce a universal and training-free framework, $\\textbf{MQM-APE}$, based on the idea of filtering out non-impactful err"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14335","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14335/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14335","created_at":"2026-07-05T09:49:10.453189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14335v2","created_at":"2026-07-05T09:49:10.453189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14335","created_at":"2026-07-05T09:49:10.453189+00:00"},{"alias_kind":"pith_short_12","alias_value":"5L5TZGO4U4QO","created_at":"2026-07-05T09:49:10.453189+00:00"},{"alias_kind":"pith_short_16","alias_value":"5L5TZGO4U4QOX4RV","created_at":"2026-07-05T09:49:10.453189+00:00"},{"alias_kind":"pith_short_8","alias_value":"5L5TZGO4","created_at":"2026-07-05T09:49:10.453189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.02975","citing_title":"TQLite: Multi-LLM Jury Guided Distillation for Real-time MQM Translation Quality Evaluation","ref_index":103,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF","json":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF.json","graph_json":"https://pith.science/api/pith-number/5L5TZGO4U4QOX4RVOLKQ3WE5PF/graph.json","events_json":"https://pith.science/api/pith-number/5L5TZGO4U4QOX4RVOLKQ3WE5PF/events.json","paper":"https://pith.science/paper/5L5TZGO4"},"agent_actions":{"view_html":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF","download_json":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF.json","view_paper":"https://pith.science/paper/5L5TZGO4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14335&json=true","fetch_graph":"https://pith.science/api/pith-number/5L5TZGO4U4QOX4RVOLKQ3WE5PF/graph.json","fetch_events":"https://pith.science/api/pith-number/5L5TZGO4U4QOX4RVOLKQ3WE5PF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF/action/storage_attestation","attest_author":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF/action/author_attestation","sign_citation":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF/action/citation_signature","submit_replication":"https://pith.science/pith/5L5TZGO4U4QOX4RVOLKQ3WE5PF/action/replication_record"}},"created_at":"2026-07-05T09:49:10.453189+00:00","updated_at":"2026-07-05T09:49:10.453189+00:00"}