{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2T7ZGZTYR6FP3O4KWQ7IUMU4JX","short_pith_number":"pith:2T7ZGZTY","schema_version":"1.0","canonical_sha256":"d4ff9366788f8afdbb8ab43e8a329c4dc637570a98351590030cc5f706f44df2","source":{"kind":"arxiv","id":"2304.09388","version":1},"attestation_state":"computed","paper":{"title":"An Empirical Study of Leveraging Knowledge Distillation for Compressing Multilingual Neural Machine Translation Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Pratyush Kumar, Raj Dabre, Varun Gumma","submitted_at":"2023-04-19T02:57:55Z","abstract_excerpt":"Knowledge distillation (KD) is a well-known method for compressing neural models. However, works focusing on distilling knowledge from large multilingual neural machine translation (MNMT) models into smaller ones are practically nonexistent, despite the popularity and superiority of MNMT. This paper bridges this gap by presenting an empirical investigation of knowledge distillation for compressing MNMT models. We take Indic to English translation as a case study and demonstrate that commonly used language-agnostic and language-aware KD approaches yield models that are 4-5x smaller but also suf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.09388","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-04-19T02:57:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"90c1cddcef3e528b56e0e95baa4a44af22410ae0fc4d6836e7b198afae2bd22e","abstract_canon_sha256":"9d4e5447f663e8014d1739be3e397544eacec13fe38d8654be12617bf86370bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:02:37.855561Z","signature_b64":"suvFYlXc2bx01EsvDDib3vuxl5RISz+mzTpc4+9VhhmI7G5BrHKPElxkiYLt5Q3FDovYU/+tSbhOVi4vg3XBBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4ff9366788f8afdbb8ab43e8a329c4dc637570a98351590030cc5f706f44df2","last_reissued_at":"2026-07-05T06:02:37.855108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:02:37.855108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study of Leveraging Knowledge Distillation for Compressing Multilingual Neural Machine Translation Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Pratyush Kumar, Raj Dabre, Varun Gumma","submitted_at":"2023-04-19T02:57:55Z","abstract_excerpt":"Knowledge distillation (KD) is a well-known method for compressing neural models. However, works focusing on distilling knowledge from large multilingual neural machine translation (MNMT) models into smaller ones are practically nonexistent, despite the popularity and superiority of MNMT. This paper bridges this gap by presenting an empirical investigation of knowledge distillation for compressing MNMT models. We take Indic to English translation as a case study and demonstrate that commonly used language-agnostic and language-aware KD approaches yield models that are 4-5x smaller but also suf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.09388","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.09388/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.09388","created_at":"2026-07-05T06:02:37.855179+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.09388v1","created_at":"2026-07-05T06:02:37.855179+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.09388","created_at":"2026-07-05T06:02:37.855179+00:00"},{"alias_kind":"pith_short_12","alias_value":"2T7ZGZTYR6FP","created_at":"2026-07-05T06:02:37.855179+00:00"},{"alias_kind":"pith_short_16","alias_value":"2T7ZGZTYR6FP3O4K","created_at":"2026-07-05T06:02:37.855179+00:00"},{"alias_kind":"pith_short_8","alias_value":"2T7ZGZTY","created_at":"2026-07-05T06:02:37.855179+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX","json":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX.json","graph_json":"https://pith.science/api/pith-number/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/graph.json","events_json":"https://pith.science/api/pith-number/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/events.json","paper":"https://pith.science/paper/2T7ZGZTY"},"agent_actions":{"view_html":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX","download_json":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX.json","view_paper":"https://pith.science/paper/2T7ZGZTY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.09388&json=true","fetch_graph":"https://pith.science/api/pith-number/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/graph.json","fetch_events":"https://pith.science/api/pith-number/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/action/storage_attestation","attest_author":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/action/author_attestation","sign_citation":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/action/citation_signature","submit_replication":"https://pith.science/pith/2T7ZGZTYR6FP3O4KWQ7IUMU4JX/action/replication_record"}},"created_at":"2026-07-05T06:02:37.855179+00:00","updated_at":"2026-07-05T06:02:37.855179+00:00"}