{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RXKCABO4YDMJF4H2M2ZZKUD6RZ","short_pith_number":"pith:RXKCABO4","schema_version":"1.0","canonical_sha256":"8dd42005dcc0d892f0fa66b395507e8e6946d05645118122e07360825d2df426","source":{"kind":"arxiv","id":"2407.03658","version":1},"attestation_state":"computed","paper":{"title":"GPT-4 vs. Human Translators: A Comprehensive Evaluation of Translation Quality Across Languages, Domains, and Expertise Levels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jianhao Yan, Judy Li, Pingchuan Yan, Xianchao Zhu, Yue Zhang, Yulong Chen","submitted_at":"2024-07-04T05:58:04Z","abstract_excerpt":"This study comprehensively evaluates the translation quality of Large Language Models (LLMs), specifically GPT-4, against human translators of varying expertise levels across multiple language pairs and domains. Through carefully designed annotation rounds, we find that GPT-4 performs comparably to junior translators in terms of total errors made but lags behind medium and senior translators. We also observe the imbalanced performance across different languages and domains, with GPT-4's translation capability gradually weakening from resource-rich to resource-poor directions. In addition, we q"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03658","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-04T05:58:04Z","cross_cats_sorted":[],"title_canon_sha256":"9256f2f2043d0c16b1a8cb1510d52cd8a7bedf71f32fe0e8d22f8953f9564351","abstract_canon_sha256":"8e66e1b16b4f18493ab91ec752c4ad21ae8a67eeb5357ad4b62a5384d7c3d057"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:21.633046Z","signature_b64":"Pysw4NfIaP+DMkDb+P/tF/LG2sTXusizel/DdpBcFWRdYEjBUMIizGa2kagaUFJSVVk8UP1Aw0nzevfCbs4GDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8dd42005dcc0d892f0fa66b395507e8e6946d05645118122e07360825d2df426","last_reissued_at":"2026-07-05T08:40:21.632568Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:21.632568Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GPT-4 vs. Human Translators: A Comprehensive Evaluation of Translation Quality Across Languages, Domains, and Expertise Levels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jianhao Yan, Judy Li, Pingchuan Yan, Xianchao Zhu, Yue Zhang, Yulong Chen","submitted_at":"2024-07-04T05:58:04Z","abstract_excerpt":"This study comprehensively evaluates the translation quality of Large Language Models (LLMs), specifically GPT-4, against human translators of varying expertise levels across multiple language pairs and domains. Through carefully designed annotation rounds, we find that GPT-4 performs comparably to junior translators in terms of total errors made but lags behind medium and senior translators. We also observe the imbalanced performance across different languages and domains, with GPT-4's translation capability gradually weakening from resource-rich to resource-poor directions. In addition, we q"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03658","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03658/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03658","created_at":"2026-07-05T08:40:21.632626+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03658v1","created_at":"2026-07-05T08:40:21.632626+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03658","created_at":"2026-07-05T08:40:21.632626+00:00"},{"alias_kind":"pith_short_12","alias_value":"RXKCABO4YDMJ","created_at":"2026-07-05T08:40:21.632626+00:00"},{"alias_kind":"pith_short_16","alias_value":"RXKCABO4YDMJF4H2","created_at":"2026-07-05T08:40:21.632626+00:00"},{"alias_kind":"pith_short_8","alias_value":"RXKCABO4","created_at":"2026-07-05T08:40:21.632626+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.08851","citing_title":"Cross-Lingual Attention Distillation with Personality-Informed Generative Augmentation for Multilingual Personality Recognition","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ","json":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ.json","graph_json":"https://pith.science/api/pith-number/RXKCABO4YDMJF4H2M2ZZKUD6RZ/graph.json","events_json":"https://pith.science/api/pith-number/RXKCABO4YDMJF4H2M2ZZKUD6RZ/events.json","paper":"https://pith.science/paper/RXKCABO4"},"agent_actions":{"view_html":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ","download_json":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ.json","view_paper":"https://pith.science/paper/RXKCABO4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03658&json=true","fetch_graph":"https://pith.science/api/pith-number/RXKCABO4YDMJF4H2M2ZZKUD6RZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RXKCABO4YDMJF4H2M2ZZKUD6RZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ/action/storage_attestation","attest_author":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ/action/author_attestation","sign_citation":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ/action/citation_signature","submit_replication":"https://pith.science/pith/RXKCABO4YDMJF4H2M2ZZKUD6RZ/action/replication_record"}},"created_at":"2026-07-05T08:40:21.632626+00:00","updated_at":"2026-07-05T08:40:21.632626+00:00"}