{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:2DZ56SGO4DKTLC5RNN3M7TZ45N","short_pith_number":"pith:2DZ56SGO","schema_version":"1.0","canonical_sha256":"d0f3df48cee0d5358bb16b76cfcf3ceb490ec1e6a58c891b15d5882f0ac0e077","source":{"kind":"arxiv","id":"2002.06823","version":1},"attestation_state":"computed","paper":{"title":"Incorporating BERT into Neural Machine Translation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Di He, Houqiang Li, Jinhua Zhu, Lijun Wu, Tao Qin, Tie-Yan Liu, Wengang Zhou, Yingce Xia","submitted_at":"2020-02-17T08:13:36Z","abstract_excerpt":"The recently proposed BERT has shown great power on a variety of natural language understanding tasks, such as text classification, reading comprehension, etc. However, how to effectively apply BERT to neural machine translation (NMT) lacks enough exploration. While BERT is more commonly used as fine-tuning instead of contextual embedding for downstream language understanding tasks, in NMT, our preliminary exploration of using BERT as contextual embedding is better than using for fine-tuning. This motivates us to think how to better leverage BERT for NMT along this direction. We propose a new "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.06823","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CL","submitted_at":"2020-02-17T08:13:36Z","cross_cats_sorted":[],"title_canon_sha256":"01147fbc80f507fc0b37194d15fae49507a72b92d840dd67de4b217c97307ad6","abstract_canon_sha256":"6149f3cf7e439066d792dd96b0052d69bc363f8acf498992434ca7fe507187ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:41:09.536209Z","signature_b64":"J9+qrjfd6sLzM1SfdNOz11EfK5vWvvhV255XLfich2tYtyAJ1TitdQ0foG8gg9wXxqYYplsEHJKEDIUyx6iIBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0f3df48cee0d5358bb16b76cfcf3ceb490ec1e6a58c891b15d5882f0ac0e077","last_reissued_at":"2026-07-05T00:41:09.535780Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:41:09.535780Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Incorporating BERT into Neural Machine Translation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Di He, Houqiang Li, Jinhua Zhu, Lijun Wu, Tao Qin, Tie-Yan Liu, Wengang Zhou, Yingce Xia","submitted_at":"2020-02-17T08:13:36Z","abstract_excerpt":"The recently proposed BERT has shown great power on a variety of natural language understanding tasks, such as text classification, reading comprehension, etc. However, how to effectively apply BERT to neural machine translation (NMT) lacks enough exploration. While BERT is more commonly used as fine-tuning instead of contextual embedding for downstream language understanding tasks, in NMT, our preliminary exploration of using BERT as contextual embedding is better than using for fine-tuning. This motivates us to think how to better leverage BERT for NMT along this direction. We propose a new "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.06823","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.06823/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.06823","created_at":"2026-07-05T00:41:09.535834+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.06823v1","created_at":"2026-07-05T00:41:09.535834+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.06823","created_at":"2026-07-05T00:41:09.535834+00:00"},{"alias_kind":"pith_short_12","alias_value":"2DZ56SGO4DKT","created_at":"2026-07-05T00:41:09.535834+00:00"},{"alias_kind":"pith_short_16","alias_value":"2DZ56SGO4DKTLC5R","created_at":"2026-07-05T00:41:09.535834+00:00"},{"alias_kind":"pith_short_8","alias_value":"2DZ56SGO","created_at":"2026-07-05T00:41:09.535834+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06818","citing_title":"Ad Headline Generation using Self-Critical Masked Language Model","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20369","citing_title":"CATCH-ME if you RAG: a dataset of Contextually Annotated multi-Turn Counterspeech against Hate and Misinformation Exchanges","ref_index":275,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22435","citing_title":"Assisted Counterspeech Writing at the Crossroads of Hate Speech and Misinformation","ref_index":265,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10845","citing_title":"BabelDOC: Better Layout-Preserving PDF Translation via Intermediate Representation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23627","citing_title":"Neural Grammatical Error Correction for Romanian","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N","json":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N.json","graph_json":"https://pith.science/api/pith-number/2DZ56SGO4DKTLC5RNN3M7TZ45N/graph.json","events_json":"https://pith.science/api/pith-number/2DZ56SGO4DKTLC5RNN3M7TZ45N/events.json","paper":"https://pith.science/paper/2DZ56SGO"},"agent_actions":{"view_html":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N","download_json":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N.json","view_paper":"https://pith.science/paper/2DZ56SGO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.06823&json=true","fetch_graph":"https://pith.science/api/pith-number/2DZ56SGO4DKTLC5RNN3M7TZ45N/graph.json","fetch_events":"https://pith.science/api/pith-number/2DZ56SGO4DKTLC5RNN3M7TZ45N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N/action/storage_attestation","attest_author":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N/action/author_attestation","sign_citation":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N/action/citation_signature","submit_replication":"https://pith.science/pith/2DZ56SGO4DKTLC5RNN3M7TZ45N/action/replication_record"}},"created_at":"2026-07-05T00:41:09.535834+00:00","updated_at":"2026-07-05T00:41:09.535834+00:00"}