{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:C6BFO47626NPOPKPZWGNPFCM3B","short_pith_number":"pith:C6BFO476","schema_version":"1.0","canonical_sha256":"17825773fed79af73d4fcd8cd7944cd859205a70728a525b8302c0b4680ba95f","source":{"kind":"arxiv","id":"2211.08633","version":2},"attestation_state":"computed","paper":{"title":"MT Metrics Correlate with Human Ratings of Simultaneous Speech Translation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dominik Mach\\'a\\v{c}ek, Ond\\v{r}ej Bojar, Raj Dabre","submitted_at":"2022-11-16T03:03:56Z","abstract_excerpt":"There have been several meta-evaluation studies on the correlation between human ratings and offline machine translation (MT) evaluation metrics such as BLEU, chrF2, BertScore and COMET. These metrics have been used to evaluate simultaneous speech translation (SST) but their correlations with human ratings of SST, which has been recently collected as Continuous Ratings (CR), are unclear. In this paper, we leverage the evaluations of candidate systems submitted to the English-German SST task at IWSLT 2022 and conduct an extensive correlation analysis of CR and the aforementioned metrics. Our st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.08633","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-11-16T03:03:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2c76fa6f44623c70ab118a49650d5ddd67d3d99fe0ab1f162593eafe01d9102c","abstract_canon_sha256":"954838e6d84a31c39676640b52e4bca55719860429c312367f455b76126c1ca6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:13.830857Z","signature_b64":"NjCwpS2XlUdHjJyTpuVwMoGHL8wc+hRdrQ2F65lHTNfvWUWVwxO+Z8eObKLxQNVa4Jf4khQMtS/wDoO0CufwDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"17825773fed79af73d4fcd8cd7944cd859205a70728a525b8302c0b4680ba95f","last_reissued_at":"2026-07-05T06:16:13.830285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:13.830285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MT Metrics Correlate with Human Ratings of Simultaneous Speech Translation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dominik Mach\\'a\\v{c}ek, Ond\\v{r}ej Bojar, Raj Dabre","submitted_at":"2022-11-16T03:03:56Z","abstract_excerpt":"There have been several meta-evaluation studies on the correlation between human ratings and offline machine translation (MT) evaluation metrics such as BLEU, chrF2, BertScore and COMET. These metrics have been used to evaluate simultaneous speech translation (SST) but their correlations with human ratings of SST, which has been recently collected as Continuous Ratings (CR), are unclear. In this paper, we leverage the evaluations of candidate systems submitted to the English-German SST task at IWSLT 2022 and conduct an extensive correlation analysis of CR and the aforementioned metrics. Our st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.08633","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.08633/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.08633","created_at":"2026-07-05T06:16:13.830367+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.08633v2","created_at":"2026-07-05T06:16:13.830367+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.08633","created_at":"2026-07-05T06:16:13.830367+00:00"},{"alias_kind":"pith_short_12","alias_value":"C6BFO47626NP","created_at":"2026-07-05T06:16:13.830367+00:00"},{"alias_kind":"pith_short_16","alias_value":"C6BFO47626NPOPKP","created_at":"2026-07-05T06:16:13.830367+00:00"},{"alias_kind":"pith_short_8","alias_value":"C6BFO476","created_at":"2026-07-05T06:16:13.830367+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17527","citing_title":"Seed LiveInterpret 2.0: End-to-end Simultaneous Speech-to-speech Translation with Your Voice","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B","json":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B.json","graph_json":"https://pith.science/api/pith-number/C6BFO47626NPOPKPZWGNPFCM3B/graph.json","events_json":"https://pith.science/api/pith-number/C6BFO47626NPOPKPZWGNPFCM3B/events.json","paper":"https://pith.science/paper/C6BFO476"},"agent_actions":{"view_html":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B","download_json":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B.json","view_paper":"https://pith.science/paper/C6BFO476","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.08633&json=true","fetch_graph":"https://pith.science/api/pith-number/C6BFO47626NPOPKPZWGNPFCM3B/graph.json","fetch_events":"https://pith.science/api/pith-number/C6BFO47626NPOPKPZWGNPFCM3B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B/action/storage_attestation","attest_author":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B/action/author_attestation","sign_citation":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B/action/citation_signature","submit_replication":"https://pith.science/pith/C6BFO47626NPOPKPZWGNPFCM3B/action/replication_record"}},"created_at":"2026-07-05T06:16:13.830367+00:00","updated_at":"2026-07-05T06:16:13.830367+00:00"}