{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GLBPTNFOBOWU727W43MYDDK77H","short_pith_number":"pith:GLBPTNFO","schema_version":"1.0","canonical_sha256":"32c2f9b4ae0bad4febf6e6d9818d5ff9d1a960212e91a44aff1979a202472dde","source":{"kind":"arxiv","id":"2503.16158","version":1},"attestation_state":"computed","paper":{"title":"Automatically Generating Chinese Homophone Words to Probe Machine Translation Estimation Systems","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Constantin Or\\u{a}san, Diptesh Kanojia, F\\'elix do Carmo, Shenbin Qian","submitted_at":"2025-03-20T13:56:15Z","abstract_excerpt":"Evaluating machine translation (MT) of user-generated content (UGC) involves unique challenges such as checking whether the nuance of emotions from the source are preserved in the target text. Recent studies have proposed emotion-related datasets, frameworks and models to automatically evaluate MT quality of Chinese UGC, without relying on reference translations. However, whether these models are robust to the challenge of preserving emotional nuances has been left largely unexplored. To address this gap, we introduce a novel method inspired by information theory which generates challenging Ch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.16158","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-20T13:56:15Z","cross_cats_sorted":[],"title_canon_sha256":"9c50a01e66d05145e2900e524869d99a2cb93972903eca2306043b3010ca8909","abstract_canon_sha256":"c2d1a6d18bc223cd5a82f200f8912fc362bd8a8cbb5e18f1eb815f2e581c868e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:36:06.884764Z","signature_b64":"EphhGBoNlQiG8jxYKi79/K8fLT93WS4W8EP47HJ2e3OUQGROAvgQAuWTVLqH7Ppn0DvlbwRoe8BTHVn+xCM3CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32c2f9b4ae0bad4febf6e6d9818d5ff9d1a960212e91a44aff1979a202472dde","last_reissued_at":"2026-07-05T10:36:06.884155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:36:06.884155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automatically Generating Chinese Homophone Words to Probe Machine Translation Estimation Systems","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Constantin Or\\u{a}san, Diptesh Kanojia, F\\'elix do Carmo, Shenbin Qian","submitted_at":"2025-03-20T13:56:15Z","abstract_excerpt":"Evaluating machine translation (MT) of user-generated content (UGC) involves unique challenges such as checking whether the nuance of emotions from the source are preserved in the target text. Recent studies have proposed emotion-related datasets, frameworks and models to automatically evaluate MT quality of Chinese UGC, without relying on reference translations. However, whether these models are robust to the challenge of preserving emotional nuances has been left largely unexplored. To address this gap, we introduce a novel method inspired by information theory which generates challenging Ch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.16158","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.16158/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.16158","created_at":"2026-07-05T10:36:06.884222+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.16158v1","created_at":"2026-07-05T10:36:06.884222+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.16158","created_at":"2026-07-05T10:36:06.884222+00:00"},{"alias_kind":"pith_short_12","alias_value":"GLBPTNFOBOWU","created_at":"2026-07-05T10:36:06.884222+00:00"},{"alias_kind":"pith_short_16","alias_value":"GLBPTNFOBOWU727W","created_at":"2026-07-05T10:36:06.884222+00:00"},{"alias_kind":"pith_short_8","alias_value":"GLBPTNFO","created_at":"2026-07-05T10:36:06.884222+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.03577","citing_title":"Looking under the Wrong Lamppost: On the Limitations of Automated Translation Quality Estimation","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H","json":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H.json","graph_json":"https://pith.science/api/pith-number/GLBPTNFOBOWU727W43MYDDK77H/graph.json","events_json":"https://pith.science/api/pith-number/GLBPTNFOBOWU727W43MYDDK77H/events.json","paper":"https://pith.science/paper/GLBPTNFO"},"agent_actions":{"view_html":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H","download_json":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H.json","view_paper":"https://pith.science/paper/GLBPTNFO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.16158&json=true","fetch_graph":"https://pith.science/api/pith-number/GLBPTNFOBOWU727W43MYDDK77H/graph.json","fetch_events":"https://pith.science/api/pith-number/GLBPTNFOBOWU727W43MYDDK77H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H/action/storage_attestation","attest_author":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H/action/author_attestation","sign_citation":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H/action/citation_signature","submit_replication":"https://pith.science/pith/GLBPTNFOBOWU727W43MYDDK77H/action/replication_record"}},"created_at":"2026-07-05T10:36:06.884222+00:00","updated_at":"2026-07-05T10:36:06.884222+00:00"}