{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NBIYSH2VMJJEENCRCE5JPACXSV","short_pith_number":"pith:NBIYSH2V","schema_version":"1.0","canonical_sha256":"6851891f556252423451113a97805795697a446ff172f19644d7864db24786b9","source":{"kind":"arxiv","id":"2208.05909","version":1},"attestation_state":"computed","paper":{"title":"Domain-Specific Text Generation for Machine Translation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andy Way, John D. Kelleher, Rejwanul Haque, Yasmin Moslem","submitted_at":"2022-08-11T16:22:16Z","abstract_excerpt":"Preservation of domain knowledge from the source to target is crucial in any translation workflow. It is common in the translation industry to receive highly specialized projects, where there is hardly any parallel in-domain data. In such scenarios where there is insufficient in-domain data to fine-tune Machine Translation (MT) models, producing translations that are consistent with the relevant context is challenging. In this work, we propose a novel approach to domain adaptation leveraging state-of-the-art pretrained language models (LMs) for domain-specific data augmentation for MT, simulat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.05909","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2022-08-11T16:22:16Z","cross_cats_sorted":[],"title_canon_sha256":"33de9ec73f922d76b6498d4315e2e55675f6ba4ddf03b0bda72ad2de0fcb3bb5","abstract_canon_sha256":"f627f304f165c009cb8eb3c9e7fb161ca588153943e95ab24437f1ab9089a9e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:57:26.875196Z","signature_b64":"p21D3Po+TdGYEPdo9yR9Lt7FK2GQjIebJH93w/5eSf9rNC/Jue2ptfdPZRu1bv/Y95+kao+yNuzogpi5eDDADg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6851891f556252423451113a97805795697a446ff172f19644d7864db24786b9","last_reissued_at":"2026-07-05T04:57:26.874767Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:57:26.874767Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Domain-Specific Text Generation for Machine Translation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andy Way, John D. Kelleher, Rejwanul Haque, Yasmin Moslem","submitted_at":"2022-08-11T16:22:16Z","abstract_excerpt":"Preservation of domain knowledge from the source to target is crucial in any translation workflow. It is common in the translation industry to receive highly specialized projects, where there is hardly any parallel in-domain data. In such scenarios where there is insufficient in-domain data to fine-tune Machine Translation (MT) models, producing translations that are consistent with the relevant context is challenging. In this work, we propose a novel approach to domain adaptation leveraging state-of-the-art pretrained language models (LMs) for domain-specific data augmentation for MT, simulat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.05909","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.05909/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.05909","created_at":"2026-07-05T04:57:26.874830+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.05909v1","created_at":"2026-07-05T04:57:26.874830+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.05909","created_at":"2026-07-05T04:57:26.874830+00:00"},{"alias_kind":"pith_short_12","alias_value":"NBIYSH2VMJJE","created_at":"2026-07-05T04:57:26.874830+00:00"},{"alias_kind":"pith_short_16","alias_value":"NBIYSH2VMJJEENCR","created_at":"2026-07-05T04:57:26.874830+00:00"},{"alias_kind":"pith_short_8","alias_value":"NBIYSH2V","created_at":"2026-07-05T04:57:26.874830+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV","json":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV.json","graph_json":"https://pith.science/api/pith-number/NBIYSH2VMJJEENCRCE5JPACXSV/graph.json","events_json":"https://pith.science/api/pith-number/NBIYSH2VMJJEENCRCE5JPACXSV/events.json","paper":"https://pith.science/paper/NBIYSH2V"},"agent_actions":{"view_html":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV","download_json":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV.json","view_paper":"https://pith.science/paper/NBIYSH2V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.05909&json=true","fetch_graph":"https://pith.science/api/pith-number/NBIYSH2VMJJEENCRCE5JPACXSV/graph.json","fetch_events":"https://pith.science/api/pith-number/NBIYSH2VMJJEENCRCE5JPACXSV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV/action/storage_attestation","attest_author":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV/action/author_attestation","sign_citation":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV/action/citation_signature","submit_replication":"https://pith.science/pith/NBIYSH2VMJJEENCRCE5JPACXSV/action/replication_record"}},"created_at":"2026-07-05T04:57:26.874830+00:00","updated_at":"2026-07-05T04:57:26.874830+00:00"}