{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ET5KI6DWE2ZONQB6JLKGJVDP2X","short_pith_number":"pith:ET5KI6DW","schema_version":"1.0","canonical_sha256":"24faa4787626b2e6c03e4ad464d46fd5f4e1959d1f01fe03e20ae3863b60017d","source":{"kind":"arxiv","id":"2401.16313","version":1},"attestation_state":"computed","paper":{"title":"Machine Translation Meta Evaluation through Translation Accuracy Challenge Sets","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandra Birch, Arnisa Fazla, Chantal Amrhein, Liane Guillou, Mark Steedman, Nikita Moghe, Rico Sennrich, Tom Kocmi","submitted_at":"2024-01-29T17:17:42Z","abstract_excerpt":"Recent machine translation (MT) metrics calibrate their effectiveness by correlating with human judgement but without any insights about their behaviour across different error types. Challenge sets are used to probe specific dimensions of metric behaviour but there are very few such datasets and they either focus on a limited number of phenomena or a limited number of language pairs. We introduce ACES, a contrastive challenge set spanning 146 language pairs, aimed at discovering whether metrics can identify 68 translation accuracy errors. These phenomena range from simple alterations at the wo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.16313","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T17:17:42Z","cross_cats_sorted":[],"title_canon_sha256":"516311241a2721cbc510d17ba5bf449479529134dae73ad576719f3de423b768","abstract_canon_sha256":"9d667326bd1e8cab70e741000a537b00ea95f3cf00da57d20a739550b20991ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:38:47.408251Z","signature_b64":"saym0wHvKc2KZnPWmy1iJQfa6IEe884Axt8svKxsE7vanoKhqGfUsO//VJzO3peRt3HN2qiEbmgGL5z8NdBLBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24faa4787626b2e6c03e4ad464d46fd5f4e1959d1f01fe03e20ae3863b60017d","last_reissued_at":"2026-07-05T07:38:47.407766Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:38:47.407766Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Machine Translation Meta Evaluation through Translation Accuracy Challenge Sets","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandra Birch, Arnisa Fazla, Chantal Amrhein, Liane Guillou, Mark Steedman, Nikita Moghe, Rico Sennrich, Tom Kocmi","submitted_at":"2024-01-29T17:17:42Z","abstract_excerpt":"Recent machine translation (MT) metrics calibrate their effectiveness by correlating with human judgement but without any insights about their behaviour across different error types. Challenge sets are used to probe specific dimensions of metric behaviour but there are very few such datasets and they either focus on a limited number of phenomena or a limited number of language pairs. We introduce ACES, a contrastive challenge set spanning 146 language pairs, aimed at discovering whether metrics can identify 68 translation accuracy errors. These phenomena range from simple alterations at the wo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.16313","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.16313/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.16313","created_at":"2026-07-05T07:38:47.407816+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.16313v1","created_at":"2026-07-05T07:38:47.407816+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.16313","created_at":"2026-07-05T07:38:47.407816+00:00"},{"alias_kind":"pith_short_12","alias_value":"ET5KI6DWE2ZO","created_at":"2026-07-05T07:38:47.407816+00:00"},{"alias_kind":"pith_short_16","alias_value":"ET5KI6DWE2ZONQB6","created_at":"2026-07-05T07:38:47.407816+00:00"},{"alias_kind":"pith_short_8","alias_value":"ET5KI6DW","created_at":"2026-07-05T07:38:47.407816+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.14909","citing_title":"Preliminary Ranking of WMT25 General Machine Translation Systems","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X","json":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X.json","graph_json":"https://pith.science/api/pith-number/ET5KI6DWE2ZONQB6JLKGJVDP2X/graph.json","events_json":"https://pith.science/api/pith-number/ET5KI6DWE2ZONQB6JLKGJVDP2X/events.json","paper":"https://pith.science/paper/ET5KI6DW"},"agent_actions":{"view_html":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X","download_json":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X.json","view_paper":"https://pith.science/paper/ET5KI6DW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.16313&json=true","fetch_graph":"https://pith.science/api/pith-number/ET5KI6DWE2ZONQB6JLKGJVDP2X/graph.json","fetch_events":"https://pith.science/api/pith-number/ET5KI6DWE2ZONQB6JLKGJVDP2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X/action/storage_attestation","attest_author":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X/action/author_attestation","sign_citation":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X/action/citation_signature","submit_replication":"https://pith.science/pith/ET5KI6DWE2ZONQB6JLKGJVDP2X/action/replication_record"}},"created_at":"2026-07-05T07:38:47.407816+00:00","updated_at":"2026-07-05T07:38:47.407816+00:00"}