{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:BONFABC74TJ55RBXNAOIGOT7OH","short_pith_number":"pith:BONFABC7","schema_version":"1.0","canonical_sha256":"0b9a50045fe4d3dec437681c833a7f71effbe24ff1125fec15e8a366eba3d5c5","source":{"kind":"arxiv","id":"2212.02800","version":1},"attestation_state":"computed","paper":{"title":"Life-long Learning for Multilingual Neural Machine Translation with Knowledge Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengqing Zong, Feifei Zhai, Jiajun Zhang, Junnan Zhu, Lu Xiang, Yang Zhao, Yu Zhou","submitted_at":"2022-12-06T07:36:16Z","abstract_excerpt":"A common scenario of Multilingual Neural Machine Translation (MNMT) is that each translation task arrives in a sequential manner, and the training data of previous tasks is unavailable. In this scenario, the current methods suffer heavily from catastrophic forgetting (CF). To alleviate the CF, we investigate knowledge distillation based life-long learning methods. Specifically, in one-tomany scenario, we propose a multilingual distillation method to make the new model (student) jointly learn multilingual output from old model (teacher) and new task. In many-to one scenario, we find that direct"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.02800","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-12-06T07:36:16Z","cross_cats_sorted":[],"title_canon_sha256":"39cfb2015863df894ae6003e80ac8ab66139146e15bc9eae8cd06fc46786c14c","abstract_canon_sha256":"a2fb3c10b3b0804e1ba1acb39a10e99cf104e12c324868cb9f063ec55a22eb46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:22:34.490382Z","signature_b64":"fC+4qPGcLRLDzC9D1nYAdDeJzBzY3gqbQHE4pY5WydQdidoWO4G2WnD9HIRXgB6MdsklEhzLaLwidIQKdvTPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b9a50045fe4d3dec437681c833a7f71effbe24ff1125fec15e8a366eba3d5c5","last_reissued_at":"2026-07-05T05:22:34.489463Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:22:34.489463Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Life-long Learning for Multilingual Neural Machine Translation with Knowledge Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengqing Zong, Feifei Zhai, Jiajun Zhang, Junnan Zhu, Lu Xiang, Yang Zhao, Yu Zhou","submitted_at":"2022-12-06T07:36:16Z","abstract_excerpt":"A common scenario of Multilingual Neural Machine Translation (MNMT) is that each translation task arrives in a sequential manner, and the training data of previous tasks is unavailable. In this scenario, the current methods suffer heavily from catastrophic forgetting (CF). To alleviate the CF, we investigate knowledge distillation based life-long learning methods. Specifically, in one-tomany scenario, we propose a multilingual distillation method to make the new model (student) jointly learn multilingual output from old model (teacher) and new task. In many-to one scenario, we find that direct"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.02800","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.02800/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.02800","created_at":"2026-07-05T05:22:34.489854+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.02800v1","created_at":"2026-07-05T05:22:34.489854+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.02800","created_at":"2026-07-05T05:22:34.489854+00:00"},{"alias_kind":"pith_short_12","alias_value":"BONFABC74TJ5","created_at":"2026-07-05T05:22:34.489854+00:00"},{"alias_kind":"pith_short_16","alias_value":"BONFABC74TJ55RBX","created_at":"2026-07-05T05:22:34.489854+00:00"},{"alias_kind":"pith_short_8","alias_value":"BONFABC7","created_at":"2026-07-05T05:22:34.489854+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12388","citing_title":"Group then Scale: Dynamic Mixture-of-Experts Multilingual Language Model","ref_index":61,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH","json":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH.json","graph_json":"https://pith.science/api/pith-number/BONFABC74TJ55RBXNAOIGOT7OH/graph.json","events_json":"https://pith.science/api/pith-number/BONFABC74TJ55RBXNAOIGOT7OH/events.json","paper":"https://pith.science/paper/BONFABC7"},"agent_actions":{"view_html":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH","download_json":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH.json","view_paper":"https://pith.science/paper/BONFABC7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.02800&json=true","fetch_graph":"https://pith.science/api/pith-number/BONFABC74TJ55RBXNAOIGOT7OH/graph.json","fetch_events":"https://pith.science/api/pith-number/BONFABC74TJ55RBXNAOIGOT7OH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH/action/storage_attestation","attest_author":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH/action/author_attestation","sign_citation":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH/action/citation_signature","submit_replication":"https://pith.science/pith/BONFABC74TJ55RBXNAOIGOT7OH/action/replication_record"}},"created_at":"2026-07-05T05:22:34.489854+00:00","updated_at":"2026-07-05T05:22:34.489854+00:00"}