{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:3ZUV4M2UX2Z62LJCWJ42XVJXQR","short_pith_number":"pith:3ZUV4M2U","schema_version":"1.0","canonical_sha256":"de695e3354beb3ed2d22b279abd537845c2984cdb069468a142d5dae150ff997","source":{"kind":"arxiv","id":"2012.14583","version":2},"attestation_state":"computed","paper":{"title":"Understanding and Improving Lexical Choice in Non-Autoregressive Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Derek F. Wong, Liang Ding, Longyue Wang, Xuebo Liu, Zhaopeng Tu","submitted_at":"2020-12-29T03:18:50Z","abstract_excerpt":"Knowledge distillation (KD) is essential for training non-autoregressive translation (NAT) models by reducing the complexity of the raw data with an autoregressive teacher model. In this study, we empirically show that as a side effect of this training, the lexical choice errors on low-frequency words are propagated to the NAT model from the teacher model. To alleviate this problem, we propose to expose the raw data to NAT models to restore the useful information of low-frequency words, which are missed in the distilled data. To this end, we introduce an extra Kullback-Leibler divergence term "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.14583","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-12-29T03:18:50Z","cross_cats_sorted":[],"title_canon_sha256":"ac1ca2e2d79904f4d8dc0dee5bb041355a2f26603cc1a8a3e054d82b36ef05e8","abstract_canon_sha256":"0c873d2c9c33b710f6ddc6b0560bda415dc510267a937c99109c72511a8dc2a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:10:07.580356Z","signature_b64":"KUYRWFkFLp9fvqahqaBASyvYfUPPfm2poWoiH4P1g7vPON9H87SMhIqOiQnngAzHBmP4lna1TnKE24HxpMuxBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de695e3354beb3ed2d22b279abd537845c2984cdb069468a142d5dae150ff997","last_reissued_at":"2026-07-05T02:10:07.579999Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:10:07.579999Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding and Improving Lexical Choice in Non-Autoregressive Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Derek F. Wong, Liang Ding, Longyue Wang, Xuebo Liu, Zhaopeng Tu","submitted_at":"2020-12-29T03:18:50Z","abstract_excerpt":"Knowledge distillation (KD) is essential for training non-autoregressive translation (NAT) models by reducing the complexity of the raw data with an autoregressive teacher model. In this study, we empirically show that as a side effect of this training, the lexical choice errors on low-frequency words are propagated to the NAT model from the teacher model. To alleviate this problem, we propose to expose the raw data to NAT models to restore the useful information of low-frequency words, which are missed in the distilled data. To this end, we introduce an extra Kullback-Leibler divergence term "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.14583","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.14583/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.14583","created_at":"2026-07-05T02:10:07.580054+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.14583v2","created_at":"2026-07-05T02:10:07.580054+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.14583","created_at":"2026-07-05T02:10:07.580054+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZUV4M2UX2Z6","created_at":"2026-07-05T02:10:07.580054+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZUV4M2UX2Z62LJC","created_at":"2026-07-05T02:10:07.580054+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZUV4M2U","created_at":"2026-07-05T02:10:07.580054+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.20156","citing_title":"Distilled Transformers with Locally Enhanced Global Representations for Face Forgery Detection","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR","json":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR.json","graph_json":"https://pith.science/api/pith-number/3ZUV4M2UX2Z62LJCWJ42XVJXQR/graph.json","events_json":"https://pith.science/api/pith-number/3ZUV4M2UX2Z62LJCWJ42XVJXQR/events.json","paper":"https://pith.science/paper/3ZUV4M2U"},"agent_actions":{"view_html":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR","download_json":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR.json","view_paper":"https://pith.science/paper/3ZUV4M2U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.14583&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZUV4M2UX2Z62LJCWJ42XVJXQR/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZUV4M2UX2Z62LJCWJ42XVJXQR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR/action/storage_attestation","attest_author":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR/action/author_attestation","sign_citation":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR/action/citation_signature","submit_replication":"https://pith.science/pith/3ZUV4M2UX2Z62LJCWJ42XVJXQR/action/replication_record"}},"created_at":"2026-07-05T02:10:07.580054+00:00","updated_at":"2026-07-05T02:10:07.580054+00:00"}