{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:5FPPGDQ3HZTHZCZX3UDSFYFLFG","short_pith_number":"pith:5FPPGDQ3","schema_version":"1.0","canonical_sha256":"e95ef30e1b3e667c8b37dd0722e0ab29b74a5f303b76b76b7ef20721ce00ca24","source":{"kind":"arxiv","id":"1909.12406","version":1},"attestation_state":"computed","paper":{"title":"Monotonic Multihead Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"James Cross, Jiatao Gu, Juan Pino, Liezl Puzon, Xutai Ma","submitted_at":"2019-09-26T21:32:50Z","abstract_excerpt":"Simultaneous machine translation models start generating a target sequence before they have encoded or read the source sequence. Recent approaches for this task either apply a fixed policy on a state-of-the art Transformer model, or a learnable monotonic attention on a weaker recurrent neural network-based structure. In this paper, we propose a new attention mechanism, Monotonic Multihead Attention (MMA), which extends the monotonic attention mechanism to multihead attention. We also introduce two novel and interpretable approaches for latency control that are specifically designed for multipl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.12406","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-09-26T21:32:50Z","cross_cats_sorted":[],"title_canon_sha256":"eb488300593bb6a0a22be9a29a0d77964b957ca2f859f318361b9ba1a3771766","abstract_canon_sha256":"59c8d918f6a0b7de631cab7d7bd3cb06cb8f8ef5141ebc7e2a1673c11e1cee59"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:07:39.333883Z","signature_b64":"vjXu97KT1/b+sXrVmHEAdag6iXNkb9pEdlRwYfYtVGCLq2Y8m7e8om25o9+ZooNcLbaAlsGbMDiavo2k+YP/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e95ef30e1b3e667c8b37dd0722e0ab29b74a5f303b76b76b7ef20721ce00ca24","last_reissued_at":"2026-07-05T00:07:39.333463Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:07:39.333463Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Monotonic Multihead Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"James Cross, Jiatao Gu, Juan Pino, Liezl Puzon, Xutai Ma","submitted_at":"2019-09-26T21:32:50Z","abstract_excerpt":"Simultaneous machine translation models start generating a target sequence before they have encoded or read the source sequence. Recent approaches for this task either apply a fixed policy on a state-of-the art Transformer model, or a learnable monotonic attention on a weaker recurrent neural network-based structure. In this paper, we propose a new attention mechanism, Monotonic Multihead Attention (MMA), which extends the monotonic attention mechanism to multihead attention. We also introduce two novel and interpretable approaches for latency control that are specifically designed for multipl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.12406","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.12406/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.12406","created_at":"2026-07-05T00:07:39.333525+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.12406v1","created_at":"2026-07-05T00:07:39.333525+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.12406","created_at":"2026-07-05T00:07:39.333525+00:00"},{"alias_kind":"pith_short_12","alias_value":"5FPPGDQ3HZTH","created_at":"2026-07-05T00:07:39.333525+00:00"},{"alias_kind":"pith_short_16","alias_value":"5FPPGDQ3HZTHZCZX","created_at":"2026-07-05T00:07:39.333525+00:00"},{"alias_kind":"pith_short_8","alias_value":"5FPPGDQ3","created_at":"2026-07-05T00:07:39.333525+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06582","citing_title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06582","citing_title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG","json":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG.json","graph_json":"https://pith.science/api/pith-number/5FPPGDQ3HZTHZCZX3UDSFYFLFG/graph.json","events_json":"https://pith.science/api/pith-number/5FPPGDQ3HZTHZCZX3UDSFYFLFG/events.json","paper":"https://pith.science/paper/5FPPGDQ3"},"agent_actions":{"view_html":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG","download_json":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG.json","view_paper":"https://pith.science/paper/5FPPGDQ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.12406&json=true","fetch_graph":"https://pith.science/api/pith-number/5FPPGDQ3HZTHZCZX3UDSFYFLFG/graph.json","fetch_events":"https://pith.science/api/pith-number/5FPPGDQ3HZTHZCZX3UDSFYFLFG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG/action/storage_attestation","attest_author":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG/action/author_attestation","sign_citation":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG/action/citation_signature","submit_replication":"https://pith.science/pith/5FPPGDQ3HZTHZCZX3UDSFYFLFG/action/replication_record"}},"created_at":"2026-07-05T00:07:39.333525+00:00","updated_at":"2026-07-05T00:07:39.333525+00:00"}