{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L4LSLU5H5CR2TRLFXB5GNSY5UD","short_pith_number":"pith:L4LSLU5H","schema_version":"1.0","canonical_sha256":"5f1725d3a7e8a3a9c565b87a66cb1da0f2a0dd8346922959009c0e69035339c1","source":{"kind":"arxiv","id":"2502.03382","version":2},"attestation_state":"computed","paper":{"title":"High-Fidelity Simultaneous Speech-To-Speech Translation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Alexandre D\\'efossez, Edouard Grave, Laurent Mazar\\'e, Neil Zeghidour, Patrick P\\'erez, Tom Labiausse","submitted_at":"2025-02-05T17:18:55Z","abstract_excerpt":"We introduce Hibiki, a decoder-only model for simultaneous speech translation. Hibiki leverages a multistream language model to synchronously process source and target speech, and jointly produces text and audio tokens to perform speech-to-text and speech-to-speech translation. We furthermore address the fundamental challenge of simultaneous interpretation, which unlike its consecutive counterpart, where one waits for the end of the source utterance to start translating, adapts its flow to accumulate just enough context to produce a correct translation in real-time, chunk by chunk. To do so, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03382","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-05T17:18:55Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"5af89aea20f5b0b10228ce6b2fed37ebec5561b463e6350baf503df9f56b992d","abstract_canon_sha256":"2f8a7630a721ff23d9a08a22e4eae91eb12659a5d5fe8d4ea9941355cf5b47eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:19.113064Z","signature_b64":"rfcIRNw3WQ9P0zlI+/GTBYvWtQKPRgWb4xT/Ei3yYMeV0Cunb+Q6eyU3ZFzMYSOqeckD+Ak1p3KEDq4+ka1vAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f1725d3a7e8a3a9c565b87a66cb1da0f2a0dd8346922959009c0e69035339c1","last_reissued_at":"2026-07-05T10:20:19.112536Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:19.112536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"High-Fidelity Simultaneous Speech-To-Speech Translation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Alexandre D\\'efossez, Edouard Grave, Laurent Mazar\\'e, Neil Zeghidour, Patrick P\\'erez, Tom Labiausse","submitted_at":"2025-02-05T17:18:55Z","abstract_excerpt":"We introduce Hibiki, a decoder-only model for simultaneous speech translation. Hibiki leverages a multistream language model to synchronously process source and target speech, and jointly produces text and audio tokens to perform speech-to-text and speech-to-speech translation. We furthermore address the fundamental challenge of simultaneous interpretation, which unlike its consecutive counterpart, where one waits for the end of the source utterance to start translating, adapts its flow to accumulate just enough context to produce a correct translation in real-time, chunk by chunk. To do so, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03382","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03382/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03382","created_at":"2026-07-05T10:20:19.112595+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03382v2","created_at":"2026-07-05T10:20:19.112595+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03382","created_at":"2026-07-05T10:20:19.112595+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4LSLU5H5CR2","created_at":"2026-07-05T10:20:19.112595+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4LSLU5H5CR2TRLF","created_at":"2026-07-05T10:20:19.112595+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4LSLU5H","created_at":"2026-07-05T10:20:19.112595+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03948","citing_title":"A Pocket Offline Model for Simultaneous Speech Translation as CUNI Submission to IWSLT 2026","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09916","citing_title":"Regularized Entropy Information Adaptation with Temporal-Awareness Networks for Simultaneous Speech Translation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD","json":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD.json","graph_json":"https://pith.science/api/pith-number/L4LSLU5H5CR2TRLFXB5GNSY5UD/graph.json","events_json":"https://pith.science/api/pith-number/L4LSLU5H5CR2TRLFXB5GNSY5UD/events.json","paper":"https://pith.science/paper/L4LSLU5H"},"agent_actions":{"view_html":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD","download_json":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD.json","view_paper":"https://pith.science/paper/L4LSLU5H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03382&json=true","fetch_graph":"https://pith.science/api/pith-number/L4LSLU5H5CR2TRLFXB5GNSY5UD/graph.json","fetch_events":"https://pith.science/api/pith-number/L4LSLU5H5CR2TRLFXB5GNSY5UD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD/action/storage_attestation","attest_author":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD/action/author_attestation","sign_citation":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD/action/citation_signature","submit_replication":"https://pith.science/pith/L4LSLU5H5CR2TRLFXB5GNSY5UD/action/replication_record"}},"created_at":"2026-07-05T10:20:19.112595+00:00","updated_at":"2026-07-05T10:20:19.112595+00:00"}