{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:TZLBPWQOTUAZDZJO7ZGVJJ7A63","short_pith_number":"pith:TZLBPWQO","canonical_record":{"source":{"id":"1808.05700","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-08-16T23:05:17Z","cross_cats_sorted":[],"title_canon_sha256":"78f4e0b047b018ba942c1cdf8d8392d240728e8a6b145b9d6bc251e632cd3568","abstract_canon_sha256":"0238799f675de4294c28c0c9aa9e4e976f45808c5e23d7e95ac731b0a9a0efab"},"schema_version":"1.0"},"canonical_sha256":"9e5617da0e9d0191e52efe4d54a7e0f6f2a2719dc0145de7e9063099dc61af06","source":{"kind":"arxiv","id":"1808.05700","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1808.05700","created_at":"2026-05-18T00:07:52Z"},{"alias_kind":"arxiv_version","alias_value":"1808.05700v1","created_at":"2026-05-18T00:07:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1808.05700","created_at":"2026-05-18T00:07:52Z"},{"alias_kind":"pith_short_12","alias_value":"TZLBPWQOTUAZ","created_at":"2026-05-18T12:32:56Z"},{"alias_kind":"pith_short_16","alias_value":"TZLBPWQOTUAZDZJO","created_at":"2026-05-18T12:32:56Z"},{"alias_kind":"pith_short_8","alias_value":"TZLBPWQO","created_at":"2026-05-18T12:32:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:TZLBPWQOTUAZDZJO7ZGVJJ7A63","target":"record","payload":{"canonical_record":{"source":{"id":"1808.05700","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-08-16T23:05:17Z","cross_cats_sorted":[],"title_canon_sha256":"78f4e0b047b018ba942c1cdf8d8392d240728e8a6b145b9d6bc251e632cd3568","abstract_canon_sha256":"0238799f675de4294c28c0c9aa9e4e976f45808c5e23d7e95ac731b0a9a0efab"},"schema_version":"1.0"},"canonical_sha256":"9e5617da0e9d0191e52efe4d54a7e0f6f2a2719dc0145de7e9063099dc61af06","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:07:52.446252Z","signature_b64":"csR+XcZxqFA8Jr0orwBjyIF9tF5IKnPpKGgbboR4zCdGSaVorI/aj8iPQssjuKGEEIqYtmI54T9jJD2azPrXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e5617da0e9d0191e52efe4d54a7e0f6f2a2719dc0145de7e9063099dc61af06","last_reissued_at":"2026-05-18T00:07:52.445583Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:07:52.445583Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1808.05700","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:07:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"57OJ5pc3QSrgPfPx7RDY8ZJ/DE1RHs2yI1pFn4j5eisw/hq4fzaXJgBlPCTIifMZ5/tFKFzywGFsj513gDDrCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-31T16:14:37.086251Z"},"content_sha256":"f1d281ae9f6afafa877fe6d5af84d635051b1b5fd11eac6c7466636d8ca8fed6","schema_version":"1.0","event_id":"sha256:f1d281ae9f6afafa877fe6d5af84d635051b1b5fd11eac6c7466636d8ca8fed6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:TZLBPWQOTUAZDZJO7ZGVJJ7A63","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Augmenting Statistical Machine Translation with Subword Translation of Out-of-Vocabulary Words","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jonathan May, Kevin Knight, Michael Pust, Nelson F. Liu","submitted_at":"2018-08-16T23:05:17Z","abstract_excerpt":"Most statistical machine translation systems cannot translate words that are unseen in the training data. However, humans can translate many classes of out-of-vocabulary (OOV) words (e.g., novel morphological variants, misspellings, and compounds) without context by using orthographic clues. Following this observation, we describe and evaluate several general methods for OOV translation that use only subword information. We pose the OOV translation problem as a standalone task and intrinsically evaluate our approaches on fourteen typologically diverse languages across varying resource levels. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1808.05700","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:07:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"D+bP6lWxNV/8q3H9Fhh6FMIazn7YgX2GRxPvg759fnmU5cauYUPbIvdCCYLsUha8olvdFxqB7XoGr48f/8L1Ag==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-31T16:14:37.086625Z"},"content_sha256":"bc006d9286239e421be814a665aaf7d7d05106a7290d8f0f7c5a5feba539b92d","schema_version":"1.0","event_id":"sha256:bc006d9286239e421be814a665aaf7d7d05106a7290d8f0f7c5a5feba539b92d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63/bundle.json","state_url":"https://pith.science/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-31T16:14:37Z","links":{"resolver":"https://pith.science/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63","bundle":"https://pith.science/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63/bundle.json","state":"https://pith.science/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63/state.json","well_known_bundle":"https://pith.science/.well-known/pith/TZLBPWQOTUAZDZJO7ZGVJJ7A63/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:TZLBPWQOTUAZDZJO7ZGVJJ7A63","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0238799f675de4294c28c0c9aa9e4e976f45808c5e23d7e95ac731b0a9a0efab","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-08-16T23:05:17Z","title_canon_sha256":"78f4e0b047b018ba942c1cdf8d8392d240728e8a6b145b9d6bc251e632cd3568"},"schema_version":"1.0","source":{"id":"1808.05700","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1808.05700","created_at":"2026-05-18T00:07:52Z"},{"alias_kind":"arxiv_version","alias_value":"1808.05700v1","created_at":"2026-05-18T00:07:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1808.05700","created_at":"2026-05-18T00:07:52Z"},{"alias_kind":"pith_short_12","alias_value":"TZLBPWQOTUAZ","created_at":"2026-05-18T12:32:56Z"},{"alias_kind":"pith_short_16","alias_value":"TZLBPWQOTUAZDZJO","created_at":"2026-05-18T12:32:56Z"},{"alias_kind":"pith_short_8","alias_value":"TZLBPWQO","created_at":"2026-05-18T12:32:56Z"}],"graph_snapshots":[{"event_id":"sha256:bc006d9286239e421be814a665aaf7d7d05106a7290d8f0f7c5a5feba539b92d","target":"graph","created_at":"2026-05-18T00:07:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Most statistical machine translation systems cannot translate words that are unseen in the training data. However, humans can translate many classes of out-of-vocabulary (OOV) words (e.g., novel morphological variants, misspellings, and compounds) without context by using orthographic clues. Following this observation, we describe and evaluate several general methods for OOV translation that use only subword information. We pose the OOV translation problem as a standalone task and intrinsically evaluate our approaches on fourteen typologically diverse languages across varying resource levels. ","authors_text":"Jonathan May, Kevin Knight, Michael Pust, Nelson F. Liu","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-08-16T23:05:17Z","title":"Augmenting Statistical Machine Translation with Subword Translation of Out-of-Vocabulary Words"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1808.05700","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f1d281ae9f6afafa877fe6d5af84d635051b1b5fd11eac6c7466636d8ca8fed6","target":"record","created_at":"2026-05-18T00:07:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0238799f675de4294c28c0c9aa9e4e976f45808c5e23d7e95ac731b0a9a0efab","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-08-16T23:05:17Z","title_canon_sha256":"78f4e0b047b018ba942c1cdf8d8392d240728e8a6b145b9d6bc251e632cd3568"},"schema_version":"1.0","source":{"id":"1808.05700","kind":"arxiv","version":1}},"canonical_sha256":"9e5617da0e9d0191e52efe4d54a7e0f6f2a2719dc0145de7e9063099dc61af06","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9e5617da0e9d0191e52efe4d54a7e0f6f2a2719dc0145de7e9063099dc61af06","first_computed_at":"2026-05-18T00:07:52.445583Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:07:52.445583Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"csR+XcZxqFA8Jr0orwBjyIF9tF5IKnPpKGgbboR4zCdGSaVorI/aj8iPQssjuKGEEIqYtmI54T9jJD2azPrXAA==","signature_status":"signed_v1","signed_at":"2026-05-18T00:07:52.446252Z","signed_message":"canonical_sha256_bytes"},"source_id":"1808.05700","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f1d281ae9f6afafa877fe6d5af84d635051b1b5fd11eac6c7466636d8ca8fed6","sha256:bc006d9286239e421be814a665aaf7d7d05106a7290d8f0f7c5a5feba539b92d"],"state_sha256":"6d094727605d701f13dfde301921894332fe9c5effaba0f16317daa4419cc1c0"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"l7PUmHVbYiNBVjuijn6um9RzUHUiBcQrZJUgXV2z0HoRLORot44R/aDQzQYCgwXa9Meg99hI9bBowWJilxaiCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-31T16:14:37.088709Z","bundle_sha256":"df5f03fcca987008d939b81c4e7fd744cceebca6e763a85045869d620a9e2b0d"}}