{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2014:DSIV2EPEWLUSRK6WTT7LKCEX2L","short_pith_number":"pith:DSIV2EPE","schema_version":"1.0","canonical_sha256":"1c915d11e4b2e928abd69cfeb50897d2c6eb9886a34a7f5a4e310b47c8cc2f18","source":{"kind":"arxiv","id":"1409.1257","version":2},"attestation_state":"computed","paper":{"title":"Overcoming the Curse of Sentence Length for Neural Machine Translation using Automatic Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE","stat.ML"],"primary_cat":"cs.CL","authors_text":"Bart van Merrienboer, Dzmitry Bahdanau, Jean Pouget-Abadie, Kyunghyun Cho, Yoshua Bengio","submitted_at":"2014-09-03T21:00:49Z","abstract_excerpt":"The authors of (Cho et al., 2014a) have shown that the recently introduced neural network translation systems suffer from a significant drop in translation quality when translating long sentences, unlike existing phrase-based translation systems. In this paper, we propose a way to address this issue by automatically segmenting an input sentence into phrases that can be easily translated by the neural network translation model. Once each segment has been independently translated by the neural machine translation model, the translated clauses are concatenated to form a final translation. Empiric"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1409.1257","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2014-09-03T21:00:49Z","cross_cats_sorted":["cs.LG","cs.NE","stat.ML"],"title_canon_sha256":"ad4817bb219b89479969aca5497eca712c0cbbb61a8603b8ed9d9f7db000e695","abstract_canon_sha256":"1bf7723c36142df29f6b1bda21fd7609e8d6ccd31bdf4cc0e95978ae75b8c301"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T02:40:55.386467Z","signature_b64":"jWFO4ZlMNGPbo0DhwGrs1dTIw5rbayghq/5yh/r0SqOhPP/n9B4dDs5CgkgSX+jZu3eQj+SyfaGiWHe8zHsOAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c915d11e4b2e928abd69cfeb50897d2c6eb9886a34a7f5a4e310b47c8cc2f18","last_reissued_at":"2026-05-18T02:40:55.385852Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T02:40:55.385852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Overcoming the Curse of Sentence Length for Neural Machine Translation using Automatic Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE","stat.ML"],"primary_cat":"cs.CL","authors_text":"Bart van Merrienboer, Dzmitry Bahdanau, Jean Pouget-Abadie, Kyunghyun Cho, Yoshua Bengio","submitted_at":"2014-09-03T21:00:49Z","abstract_excerpt":"The authors of (Cho et al., 2014a) have shown that the recently introduced neural network translation systems suffer from a significant drop in translation quality when translating long sentences, unlike existing phrase-based translation systems. In this paper, we propose a way to address this issue by automatically segmenting an input sentence into phrases that can be easily translated by the neural network translation model. Once each segment has been independently translated by the neural machine translation model, the translated clauses are concatenated to form a final translation. Empiric"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1409.1257","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1409.1257","created_at":"2026-05-18T02:40:55.385945+00:00"},{"alias_kind":"arxiv_version","alias_value":"1409.1257v2","created_at":"2026-05-18T02:40:55.385945+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1409.1257","created_at":"2026-05-18T02:40:55.385945+00:00"},{"alias_kind":"pith_short_12","alias_value":"DSIV2EPEWLUS","created_at":"2026-05-18T12:28:25.294606+00:00"},{"alias_kind":"pith_short_16","alias_value":"DSIV2EPEWLUSRK6W","created_at":"2026-05-18T12:28:25.294606+00:00"},{"alias_kind":"pith_short_8","alias_value":"DSIV2EPE","created_at":"2026-05-18T12:28:25.294606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.19604","citing_title":"Evaluating Machine Translation Models for English-Hindi Language Pairs: A Comparative Analysis","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L","json":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L.json","graph_json":"https://pith.science/api/pith-number/DSIV2EPEWLUSRK6WTT7LKCEX2L/graph.json","events_json":"https://pith.science/api/pith-number/DSIV2EPEWLUSRK6WTT7LKCEX2L/events.json","paper":"https://pith.science/paper/DSIV2EPE"},"agent_actions":{"view_html":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L","download_json":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L.json","view_paper":"https://pith.science/paper/DSIV2EPE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1409.1257&json=true","fetch_graph":"https://pith.science/api/pith-number/DSIV2EPEWLUSRK6WTT7LKCEX2L/graph.json","fetch_events":"https://pith.science/api/pith-number/DSIV2EPEWLUSRK6WTT7LKCEX2L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L/action/storage_attestation","attest_author":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L/action/author_attestation","sign_citation":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L/action/citation_signature","submit_replication":"https://pith.science/pith/DSIV2EPEWLUSRK6WTT7LKCEX2L/action/replication_record"}},"created_at":"2026-05-18T02:40:55.385945+00:00","updated_at":"2026-05-18T02:40:55.385945+00:00"}