{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7LFRH53BPNO7B5KUOKRJGO3UON","short_pith_number":"pith:7LFRH53B","schema_version":"1.0","canonical_sha256":"facb13f7617b5df0f55472a2933b7473520189a0bbc88db2c8471d897ac8317d","source":{"kind":"arxiv","id":"2408.09430","version":1},"attestation_state":"computed","paper":{"title":"FASST: Fast LLM-based Simultaneous Speech Translation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chinmay Dandekar, Lei Li, Siqi Ouyang, Xi Xu","submitted_at":"2024-08-18T10:12:39Z","abstract_excerpt":"Simultaneous speech translation (SST) takes streaming speech input and generates text translation on the fly. Existing methods either have high latency due to recomputation of input representations, or fall behind of offline ST in translation quality. In this paper, we propose FASST, a fast large language model based method for streaming speech translation. We propose blockwise-causal speech encoding and consistency mask, so that streaming speech input can be encoded incrementally without recomputation. Furthermore, we develop a two-stage training strategy to optimize FASST for simultaneous in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.09430","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-18T10:12:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"40fd8b7a3ba8894dbcda3a968d5a54635f01adf7d7fae09bb9e35bea971789be","abstract_canon_sha256":"ff0d1f10f95bd2ac23fb84ce1634f61d84a29f9af3b7899357a2c4ecef7a3c81"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:32.367088Z","signature_b64":"/WLUjWK60wiPFWlLQrLBqgxggbfLICyKVyTH2mF0wlY4k+XxYnpGpJr3+717D6I+4eAQQuF5VTegZpncFboRBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"facb13f7617b5df0f55472a2933b7473520189a0bbc88db2c8471d897ac8317d","last_reissued_at":"2026-07-05T08:56:32.366661Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:32.366661Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FASST: Fast LLM-based Simultaneous Speech Translation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chinmay Dandekar, Lei Li, Siqi Ouyang, Xi Xu","submitted_at":"2024-08-18T10:12:39Z","abstract_excerpt":"Simultaneous speech translation (SST) takes streaming speech input and generates text translation on the fly. Existing methods either have high latency due to recomputation of input representations, or fall behind of offline ST in translation quality. In this paper, we propose FASST, a fast large language model based method for streaming speech translation. We propose blockwise-causal speech encoding and consistency mask, so that streaming speech input can be encoded incrementally without recomputation. Furthermore, we develop a two-stage training strategy to optimize FASST for simultaneous in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.09430","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.09430/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.09430","created_at":"2026-07-05T08:56:32.366717+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.09430v1","created_at":"2026-07-05T08:56:32.366717+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.09430","created_at":"2026-07-05T08:56:32.366717+00:00"},{"alias_kind":"pith_short_12","alias_value":"7LFRH53BPNO7","created_at":"2026-07-05T08:56:32.366717+00:00"},{"alias_kind":"pith_short_16","alias_value":"7LFRH53BPNO7B5KU","created_at":"2026-07-05T08:56:32.366717+00:00"},{"alias_kind":"pith_short_8","alias_value":"7LFRH53B","created_at":"2026-07-05T08:56:32.366717+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31432","citing_title":"DOA: Training-Free Decoder-Only Attention Policy for Long-Form Simultaneous Translation with SpeechLLMs","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON","json":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON.json","graph_json":"https://pith.science/api/pith-number/7LFRH53BPNO7B5KUOKRJGO3UON/graph.json","events_json":"https://pith.science/api/pith-number/7LFRH53BPNO7B5KUOKRJGO3UON/events.json","paper":"https://pith.science/paper/7LFRH53B"},"agent_actions":{"view_html":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON","download_json":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON.json","view_paper":"https://pith.science/paper/7LFRH53B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.09430&json=true","fetch_graph":"https://pith.science/api/pith-number/7LFRH53BPNO7B5KUOKRJGO3UON/graph.json","fetch_events":"https://pith.science/api/pith-number/7LFRH53BPNO7B5KUOKRJGO3UON/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON/action/storage_attestation","attest_author":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON/action/author_attestation","sign_citation":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON/action/citation_signature","submit_replication":"https://pith.science/pith/7LFRH53BPNO7B5KUOKRJGO3UON/action/replication_record"}},"created_at":"2026-07-05T08:56:32.366717+00:00","updated_at":"2026-07-05T08:56:32.366717+00:00"}