{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SW4UW57NAM7IL6WG32LOEIJM25","short_pith_number":"pith:SW4UW57N","schema_version":"1.0","canonical_sha256":"95b94b77ed033e85fac6de96e2212cd754c19912d951ece814c785b3f4825071","source":{"kind":"arxiv","id":"2410.18697","version":2},"attestation_state":"computed","paper":{"title":"How Good Are LLMs for Literary Translation, Really? Literary Translation Evaluation with Humans and LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ran Zhang, Steffen Eger, Wei Zhao","submitted_at":"2024-10-24T12:48:03Z","abstract_excerpt":"Recent research has focused on literary machine translation (MT) as a new challenge in MT. However, the evaluation of literary MT remains an open problem. We contribute to this ongoing discussion by introducing LITEVAL-CORPUS, a paragraph-level parallel corpus containing verified human translations and outputs from 9 MT systems, which totals over 2k translations and 13k evaluated sentences across four language pairs, costing 4.5k C. This corpus enables us to (i) examine the consistency and adequacy of human evaluation schemes with various degrees of complexity, (ii) compare evaluations by stud"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18697","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T12:48:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"92ca75dcc2a3c12fb528e233b249b07970eb14edf01fd36affc1ee907e17d56b","abstract_canon_sha256":"a6fbe0352c454a261e03b7c9f8d18e4c995bb5952e6f89d9682a6a58f04797b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:30.164202Z","signature_b64":"SsREPUVKv21ertpzFlmHh1z4jY+LZ5799Nv0bSNhRn6TxeXfw5PMzN4Wr0WPE4W1SIDnYMvJmYyRxd7IlWvmCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95b94b77ed033e85fac6de96e2212cd754c19912d951ece814c785b3f4825071","last_reissued_at":"2026-07-05T10:19:30.163668Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:30.163668Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Good Are LLMs for Literary Translation, Really? Literary Translation Evaluation with Humans and LLMs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ran Zhang, Steffen Eger, Wei Zhao","submitted_at":"2024-10-24T12:48:03Z","abstract_excerpt":"Recent research has focused on literary machine translation (MT) as a new challenge in MT. However, the evaluation of literary MT remains an open problem. We contribute to this ongoing discussion by introducing LITEVAL-CORPUS, a paragraph-level parallel corpus containing verified human translations and outputs from 9 MT systems, which totals over 2k translations and 13k evaluated sentences across four language pairs, costing 4.5k C. This corpus enables us to (i) examine the consistency and adequacy of human evaluation schemes with various degrees of complexity, (ii) compare evaluations by stud"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18697","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18697/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18697","created_at":"2026-07-05T10:19:30.163725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18697v2","created_at":"2026-07-05T10:19:30.163725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18697","created_at":"2026-07-05T10:19:30.163725+00:00"},{"alias_kind":"pith_short_12","alias_value":"SW4UW57NAM7I","created_at":"2026-07-05T10:19:30.163725+00:00"},{"alias_kind":"pith_short_16","alias_value":"SW4UW57NAM7IL6WG","created_at":"2026-07-05T10:19:30.163725+00:00"},{"alias_kind":"pith_short_8","alias_value":"SW4UW57N","created_at":"2026-07-05T10:19:30.163725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.04497","citing_title":"Opportunities and Challenges of Large Language Models for Low-Resource Languages in Humanities Research","ref_index":151,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25","json":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25.json","graph_json":"https://pith.science/api/pith-number/SW4UW57NAM7IL6WG32LOEIJM25/graph.json","events_json":"https://pith.science/api/pith-number/SW4UW57NAM7IL6WG32LOEIJM25/events.json","paper":"https://pith.science/paper/SW4UW57N"},"agent_actions":{"view_html":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25","download_json":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25.json","view_paper":"https://pith.science/paper/SW4UW57N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18697&json=true","fetch_graph":"https://pith.science/api/pith-number/SW4UW57NAM7IL6WG32LOEIJM25/graph.json","fetch_events":"https://pith.science/api/pith-number/SW4UW57NAM7IL6WG32LOEIJM25/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25/action/storage_attestation","attest_author":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25/action/author_attestation","sign_citation":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25/action/citation_signature","submit_replication":"https://pith.science/pith/SW4UW57NAM7IL6WG32LOEIJM25/action/replication_record"}},"created_at":"2026-07-05T10:19:30.163725+00:00","updated_at":"2026-07-05T10:19:30.163725+00:00"}