{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:7FJ3QTQIDVRY5DKDC7WIDE4F72","short_pith_number":"pith:7FJ3QTQI","schema_version":"1.0","canonical_sha256":"f953b84e081d638e8d4317ec819385feb86fbe2cdbc4ad74e220832b4eeafad1","source":{"kind":"arxiv","id":"2105.12967","version":1},"attestation_state":"computed","paper":{"title":"Selective Knowledge Distillation for Neural Machine Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Fusheng Wang, Jianhao Yan, Jie Zhou","submitted_at":"2021-05-27T06:54:12Z","abstract_excerpt":"Neural Machine Translation (NMT) models achieve state-of-the-art performance on many translation benchmarks. As an active research field in NMT, knowledge distillation is widely applied to enhance the model's performance by transferring teacher model's knowledge on each training sample. However, previous work rarely discusses the different impacts and connections among these samples, which serve as the medium for transferring teacher knowledge. In this paper, we design a novel protocol that can effectively analyze the different impacts of samples by comparing various samples' partitions. Based"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.12967","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-05-27T06:54:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6c53ffd1c1479e2ecdd8d976d8f8a504197608c0dc35b26f7f906b037be8cea2","abstract_canon_sha256":"6b217461e1ae1bbd70cea5e0fcf89e65b5fbb3abb4025fae3b219df45265e83e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:43:55.838273Z","signature_b64":"mCYAsxq/cBBsJPGxKUf8fjP2dDumbSZbDEtt2H+brD2TJE3u3C0s1L3iOXEW54gnA2SrE/lhxuawe5hlSJjNDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f953b84e081d638e8d4317ec819385feb86fbe2cdbc4ad74e220832b4eeafad1","last_reissued_at":"2026-07-05T02:43:55.837739Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:43:55.837739Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Selective Knowledge Distillation for Neural Machine Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Fusheng Wang, Jianhao Yan, Jie Zhou","submitted_at":"2021-05-27T06:54:12Z","abstract_excerpt":"Neural Machine Translation (NMT) models achieve state-of-the-art performance on many translation benchmarks. As an active research field in NMT, knowledge distillation is widely applied to enhance the model's performance by transferring teacher model's knowledge on each training sample. However, previous work rarely discusses the different impacts and connections among these samples, which serve as the medium for transferring teacher knowledge. In this paper, we design a novel protocol that can effectively analyze the different impacts of samples by comparing various samples' partitions. Based"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.12967","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.12967/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.12967","created_at":"2026-07-05T02:43:55.837800+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.12967v1","created_at":"2026-07-05T02:43:55.837800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.12967","created_at":"2026-07-05T02:43:55.837800+00:00"},{"alias_kind":"pith_short_12","alias_value":"7FJ3QTQIDVRY","created_at":"2026-07-05T02:43:55.837800+00:00"},{"alias_kind":"pith_short_16","alias_value":"7FJ3QTQIDVRY5DKD","created_at":"2026-07-05T02:43:55.837800+00:00"},{"alias_kind":"pith_short_8","alias_value":"7FJ3QTQI","created_at":"2026-07-05T02:43:55.837800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09924","citing_title":"Evolving Knowledge Distillation for Lightweight Neural Machine Translation","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72","json":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72.json","graph_json":"https://pith.science/api/pith-number/7FJ3QTQIDVRY5DKDC7WIDE4F72/graph.json","events_json":"https://pith.science/api/pith-number/7FJ3QTQIDVRY5DKDC7WIDE4F72/events.json","paper":"https://pith.science/paper/7FJ3QTQI"},"agent_actions":{"view_html":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72","download_json":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72.json","view_paper":"https://pith.science/paper/7FJ3QTQI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.12967&json=true","fetch_graph":"https://pith.science/api/pith-number/7FJ3QTQIDVRY5DKDC7WIDE4F72/graph.json","fetch_events":"https://pith.science/api/pith-number/7FJ3QTQIDVRY5DKDC7WIDE4F72/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72/action/storage_attestation","attest_author":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72/action/author_attestation","sign_citation":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72/action/citation_signature","submit_replication":"https://pith.science/pith/7FJ3QTQIDVRY5DKDC7WIDE4F72/action/replication_record"}},"created_at":"2026-07-05T02:43:55.837800+00:00","updated_at":"2026-07-05T02:43:55.837800+00:00"}