{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:A2P5ZPXC7SBRINRXP6MDZCWSEU","short_pith_number":"pith:A2P5ZPXC","schema_version":"1.0","canonical_sha256":"069fdcbee2fc831436377f983c8ad22526d5ea6e2856b15e6d7f8e7f7eebcc5f","source":{"kind":"arxiv","id":"2012.15833","version":1},"attestation_state":"computed","paper":{"title":"Fully Non-autoregressive Neural Machine Translation: Tricks of the Trade","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiatao Gu, Xiang Kong","submitted_at":"2020-12-31T18:52:59Z","abstract_excerpt":"Fully non-autoregressive neural machine translation (NAT) is proposed to simultaneously predict tokens with single forward of neural networks, which significantly reduces the inference latency at the expense of quality drop compared to the Transformer baseline. In this work, we target on closing the performance gap while maintaining the latency advantage. We first inspect the fundamental issues of fully NAT models, and adopt dependency reduction in the learning space of output tokens as the basic guidance. Then, we revisit methods in four different aspects that have been proven effective for i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.15833","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-12-31T18:52:59Z","cross_cats_sorted":[],"title_canon_sha256":"40a84400d00345362c3235abfda67b507b1e66b3ceebfa19cceb1a5f94e865b0","abstract_canon_sha256":"c9cb41dc98717b077d6813977edd27df286255cb95472b517bfc2b8d613b300c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:03:21.828008Z","signature_b64":"N1UdH975aqinRrDgxQq3jzwlFBAT1V0c0dOH9OeevDUJ6K4t5dMM3hrAXCvXSQvDVdwhS/XzUA11v7ty3fozDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"069fdcbee2fc831436377f983c8ad22526d5ea6e2856b15e6d7f8e7f7eebcc5f","last_reissued_at":"2026-07-05T02:03:21.827593Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:03:21.827593Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fully Non-autoregressive Neural Machine Translation: Tricks of the Trade","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiatao Gu, Xiang Kong","submitted_at":"2020-12-31T18:52:59Z","abstract_excerpt":"Fully non-autoregressive neural machine translation (NAT) is proposed to simultaneously predict tokens with single forward of neural networks, which significantly reduces the inference latency at the expense of quality drop compared to the Transformer baseline. In this work, we target on closing the performance gap while maintaining the latency advantage. We first inspect the fundamental issues of fully NAT models, and adopt dependency reduction in the learning space of output tokens as the basic guidance. Then, we revisit methods in four different aspects that have been proven effective for i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.15833","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.15833/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.15833","created_at":"2026-07-05T02:03:21.827650+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.15833v1","created_at":"2026-07-05T02:03:21.827650+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.15833","created_at":"2026-07-05T02:03:21.827650+00:00"},{"alias_kind":"pith_short_12","alias_value":"A2P5ZPXC7SBR","created_at":"2026-07-05T02:03:21.827650+00:00"},{"alias_kind":"pith_short_16","alias_value":"A2P5ZPXC7SBRINRX","created_at":"2026-07-05T02:03:21.827650+00:00"},{"alias_kind":"pith_short_8","alias_value":"A2P5ZPXC","created_at":"2026-07-05T02:03:21.827650+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU","json":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU.json","graph_json":"https://pith.science/api/pith-number/A2P5ZPXC7SBRINRXP6MDZCWSEU/graph.json","events_json":"https://pith.science/api/pith-number/A2P5ZPXC7SBRINRXP6MDZCWSEU/events.json","paper":"https://pith.science/paper/A2P5ZPXC"},"agent_actions":{"view_html":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU","download_json":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU.json","view_paper":"https://pith.science/paper/A2P5ZPXC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.15833&json=true","fetch_graph":"https://pith.science/api/pith-number/A2P5ZPXC7SBRINRXP6MDZCWSEU/graph.json","fetch_events":"https://pith.science/api/pith-number/A2P5ZPXC7SBRINRXP6MDZCWSEU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU/action/storage_attestation","attest_author":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU/action/author_attestation","sign_citation":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU/action/citation_signature","submit_replication":"https://pith.science/pith/A2P5ZPXC7SBRINRXP6MDZCWSEU/action/replication_record"}},"created_at":"2026-07-05T02:03:21.827650+00:00","updated_at":"2026-07-05T02:03:21.827650+00:00"}