{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PPMYDYJ6G2OV5DJA5IXLXQRANJ","short_pith_number":"pith:PPMYDYJ6","schema_version":"1.0","canonical_sha256":"7bd981e13e369d5e8d20ea2ebbc2206a6419a9a18a5efb0ff3018bd245a0fb9e","source":{"kind":"arxiv","id":"2305.07005","version":1},"attestation_state":"computed","paper":{"title":"Subword Segmental Machine Translation: Unifying Segmentation and Target Sentence Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Francois Meyer, Jan Buys","submitted_at":"2023-05-11T17:44:29Z","abstract_excerpt":"Subword segmenters like BPE operate as a preprocessing step in neural machine translation and other (conditional) language models. They are applied to datasets before training, so translation or text generation quality relies on the quality of segmentations. We propose a departure from this paradigm, called subword segmental machine translation (SSMT). SSMT unifies subword segmentation and MT in a single trainable model. It learns to segment target sentence words while jointly learning to generate target sentences. To use SSMT during inference we propose dynamic decoding, a text generation alg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.07005","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-11T17:44:29Z","cross_cats_sorted":[],"title_canon_sha256":"d612e316c80d059ef6f9cc6423493d199e78614de1d5c6f52b9090e86ae24fb9","abstract_canon_sha256":"090386af375f47a48d7c5ab55ffa3a1b8188ec0be11a59039c8110844c87b326"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:09:18.477179Z","signature_b64":"GcZHXbi3xqfoJiMiphmPsmtiUgcLkostCzp4sIRRvGS+VpZKyTIiSszigbwrKoWgJ8xqyoHhrnX8+F3hk8wfDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7bd981e13e369d5e8d20ea2ebbc2206a6419a9a18a5efb0ff3018bd245a0fb9e","last_reissued_at":"2026-07-05T06:09:18.476771Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:09:18.476771Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Subword Segmental Machine Translation: Unifying Segmentation and Target Sentence Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Francois Meyer, Jan Buys","submitted_at":"2023-05-11T17:44:29Z","abstract_excerpt":"Subword segmenters like BPE operate as a preprocessing step in neural machine translation and other (conditional) language models. They are applied to datasets before training, so translation or text generation quality relies on the quality of segmentations. We propose a departure from this paradigm, called subword segmental machine translation (SSMT). SSMT unifies subword segmentation and MT in a single trainable model. It learns to segment target sentence words while jointly learning to generate target sentences. To use SSMT during inference we propose dynamic decoding, a text generation alg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.07005","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.07005/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.07005","created_at":"2026-07-05T06:09:18.476827+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.07005v1","created_at":"2026-07-05T06:09:18.476827+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.07005","created_at":"2026-07-05T06:09:18.476827+00:00"},{"alias_kind":"pith_short_12","alias_value":"PPMYDYJ6G2OV","created_at":"2026-07-05T06:09:18.476827+00:00"},{"alias_kind":"pith_short_16","alias_value":"PPMYDYJ6G2OV5DJA","created_at":"2026-07-05T06:09:18.476827+00:00"},{"alias_kind":"pith_short_8","alias_value":"PPMYDYJ6","created_at":"2026-07-05T06:09:18.476827+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13436","citing_title":"Pretraining Language Models with Subword Regularization: An Empirical Study of BPE Dropout in Low-Resource NLP","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ","json":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ.json","graph_json":"https://pith.science/api/pith-number/PPMYDYJ6G2OV5DJA5IXLXQRANJ/graph.json","events_json":"https://pith.science/api/pith-number/PPMYDYJ6G2OV5DJA5IXLXQRANJ/events.json","paper":"https://pith.science/paper/PPMYDYJ6"},"agent_actions":{"view_html":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ","download_json":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ.json","view_paper":"https://pith.science/paper/PPMYDYJ6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.07005&json=true","fetch_graph":"https://pith.science/api/pith-number/PPMYDYJ6G2OV5DJA5IXLXQRANJ/graph.json","fetch_events":"https://pith.science/api/pith-number/PPMYDYJ6G2OV5DJA5IXLXQRANJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ/action/storage_attestation","attest_author":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ/action/author_attestation","sign_citation":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ/action/citation_signature","submit_replication":"https://pith.science/pith/PPMYDYJ6G2OV5DJA5IXLXQRANJ/action/replication_record"}},"created_at":"2026-07-05T06:09:18.476827+00:00","updated_at":"2026-07-05T06:09:18.476827+00:00"}