{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:B7D26AXEO2WITVED3OBIPROXLB","short_pith_number":"pith:B7D26AXE","schema_version":"1.0","canonical_sha256":"0fc7af02e476ac89d483db8287c5d7586ef12a604a09f66e497a91cacc4af495","source":{"kind":"arxiv","id":"1911.08717","version":2},"attestation_state":"computed","paper":{"title":"Fine-Tuning by Curriculum Learning for Non-Autoregressive Neural Machine Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Enhong Chen, Junliang Guo, Linli Xu, Tao Qin, Tie-Yan Liu, Xu Tan","submitted_at":"2019-11-20T05:48:31Z","abstract_excerpt":"Non-autoregressive translation (NAT) models remove the dependence on previous target tokens and generate all target tokens in parallel, resulting in significant inference speedup but at the cost of inferior translation accuracy compared to autoregressive translation (AT) models. Considering that AT models have higher accuracy and are easier to train than NAT models, and both of them share the same model configurations, a natural idea to improve the accuracy of NAT models is to transfer a well-trained AT model to an NAT model through fine-tuning. However, since AT and NAT models differ greatly "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.08717","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-20T05:48:31Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"1ff4803aba74234bc095551600f0e1941caa66211d3d84b2de353b29115ab0ee","abstract_canon_sha256":"696ab82fe4aae5868e750eb4e1daccc3acff3e681b4270594c191193b26ae944"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:21:16.773404Z","signature_b64":"z8MKEQrmky6V2k0Zig/4Ylz3Dw3Nua8J2VuVxyOqGIK+laFpgzpllc9wJ/AT/6nJyfVHddnVV59lpfEGqfxGBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0fc7af02e476ac89d483db8287c5d7586ef12a604a09f66e497a91cacc4af495","last_reissued_at":"2026-07-05T00:21:16.773001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:21:16.773001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-Tuning by Curriculum Learning for Non-Autoregressive Neural Machine Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Enhong Chen, Junliang Guo, Linli Xu, Tao Qin, Tie-Yan Liu, Xu Tan","submitted_at":"2019-11-20T05:48:31Z","abstract_excerpt":"Non-autoregressive translation (NAT) models remove the dependence on previous target tokens and generate all target tokens in parallel, resulting in significant inference speedup but at the cost of inferior translation accuracy compared to autoregressive translation (AT) models. Considering that AT models have higher accuracy and are easier to train than NAT models, and both of them share the same model configurations, a natural idea to improve the accuracy of NAT models is to transfer a well-trained AT model to an NAT model through fine-tuning. However, since AT and NAT models differ greatly "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.08717","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.08717/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.08717","created_at":"2026-07-05T00:21:16.773057+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.08717v2","created_at":"2026-07-05T00:21:16.773057+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.08717","created_at":"2026-07-05T00:21:16.773057+00:00"},{"alias_kind":"pith_short_12","alias_value":"B7D26AXEO2WI","created_at":"2026-07-05T00:21:16.773057+00:00"},{"alias_kind":"pith_short_16","alias_value":"B7D26AXEO2WITVED","created_at":"2026-07-05T00:21:16.773057+00:00"},{"alias_kind":"pith_short_8","alias_value":"B7D26AXE","created_at":"2026-07-05T00:21:16.773057+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.12639","citing_title":"Falcon: Faster and Parallel Inference of Large Language Models through Enhanced Semi-Autoregressive Drafting and Custom-Designed Decoding Tree","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB","json":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB.json","graph_json":"https://pith.science/api/pith-number/B7D26AXEO2WITVED3OBIPROXLB/graph.json","events_json":"https://pith.science/api/pith-number/B7D26AXEO2WITVED3OBIPROXLB/events.json","paper":"https://pith.science/paper/B7D26AXE"},"agent_actions":{"view_html":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB","download_json":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB.json","view_paper":"https://pith.science/paper/B7D26AXE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.08717&json=true","fetch_graph":"https://pith.science/api/pith-number/B7D26AXEO2WITVED3OBIPROXLB/graph.json","fetch_events":"https://pith.science/api/pith-number/B7D26AXEO2WITVED3OBIPROXLB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB/action/storage_attestation","attest_author":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB/action/author_attestation","sign_citation":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB/action/citation_signature","submit_replication":"https://pith.science/pith/B7D26AXEO2WITVED3OBIPROXLB/action/replication_record"}},"created_at":"2026-07-05T00:21:16.773057+00:00","updated_at":"2026-07-05T00:21:16.773057+00:00"}