{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PMH7SX57LBZAH5ZVB3KNUIUFDS","short_pith_number":"pith:PMH7SX57","schema_version":"1.0","canonical_sha256":"7b0ff95fbf587203f7350ed4da22851cb918f76f89578904ed005189a913bfde","source":{"kind":"arxiv","id":"2402.12025","version":3},"attestation_state":"computed","paper":{"title":"Speech Translation with Speech Foundation Models and Large Language Models: What is There and What is Missing?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Luisa Bentivogli, Marco Gaido, Matteo Negri, Sara Papi","submitted_at":"2024-02-19T10:34:13Z","abstract_excerpt":"The field of natural language processing (NLP) has recently witnessed a transformative shift with the emergence of foundation models, particularly Large Language Models (LLMs) that have revolutionized text-based NLP. This paradigm has extended to other modalities, including speech, where researchers are actively exploring the combination of Speech Foundation Models (SFMs) and LLMs into single, unified models capable of addressing multimodal tasks. Among such tasks, this paper focuses on speech-to-text translation (ST). By examining the published papers on the topic, we propose a unified view o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12025","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-19T10:34:13Z","cross_cats_sorted":[],"title_canon_sha256":"94a528c6f044c536981578f127306cf1ac3745ca208aa01842d546cbeca3f988","abstract_canon_sha256":"66555bd057a5a55f63516db025792c50d358788307b5091a848ba31907291326"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:32.263073Z","signature_b64":"gEDzW1tswyoCBQj6U8vE5SDJZQ1Arb7RRW2tEZf19Ux3uB9+fC57lpBnE/oVTalWZ3dJNJtggWeZX9bIbwQWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b0ff95fbf587203f7350ed4da22851cb918f76f89578904ed005189a913bfde","last_reissued_at":"2026-07-05T09:41:32.262596Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:32.262596Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speech Translation with Speech Foundation Models and Large Language Models: What is There and What is Missing?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Luisa Bentivogli, Marco Gaido, Matteo Negri, Sara Papi","submitted_at":"2024-02-19T10:34:13Z","abstract_excerpt":"The field of natural language processing (NLP) has recently witnessed a transformative shift with the emergence of foundation models, particularly Large Language Models (LLMs) that have revolutionized text-based NLP. This paradigm has extended to other modalities, including speech, where researchers are actively exploring the combination of Speech Foundation Models (SFMs) and LLMs into single, unified models capable of addressing multimodal tasks. Among such tasks, this paper focuses on speech-to-text translation (ST). By examining the published papers on the topic, we propose a unified view o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12025","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12025","created_at":"2026-07-05T09:41:32.262654+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12025v3","created_at":"2026-07-05T09:41:32.262654+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12025","created_at":"2026-07-05T09:41:32.262654+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMH7SX57LBZA","created_at":"2026-07-05T09:41:32.262654+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMH7SX57LBZAH5ZV","created_at":"2026-07-05T09:41:32.262654+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMH7SX57","created_at":"2026-07-05T09:41:32.262654+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01381","citing_title":"A framework for analyzing concept representations in neural models","ref_index":241,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS","json":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS.json","graph_json":"https://pith.science/api/pith-number/PMH7SX57LBZAH5ZVB3KNUIUFDS/graph.json","events_json":"https://pith.science/api/pith-number/PMH7SX57LBZAH5ZVB3KNUIUFDS/events.json","paper":"https://pith.science/paper/PMH7SX57"},"agent_actions":{"view_html":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS","download_json":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS.json","view_paper":"https://pith.science/paper/PMH7SX57","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12025&json=true","fetch_graph":"https://pith.science/api/pith-number/PMH7SX57LBZAH5ZVB3KNUIUFDS/graph.json","fetch_events":"https://pith.science/api/pith-number/PMH7SX57LBZAH5ZVB3KNUIUFDS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS/action/storage_attestation","attest_author":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS/action/author_attestation","sign_citation":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS/action/citation_signature","submit_replication":"https://pith.science/pith/PMH7SX57LBZAH5ZVB3KNUIUFDS/action/replication_record"}},"created_at":"2026-07-05T09:41:32.262654+00:00","updated_at":"2026-07-05T09:41:32.262654+00:00"}