{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TUUE7AX5W6F3GTT3UU22GTOH56","short_pith_number":"pith:TUUE7AX5","schema_version":"1.0","canonical_sha256":"9d284f82fdb78bb34e7ba535a34dc7efbc3994707348fa17cd03dd8f6a6f8592","source":{"kind":"arxiv","id":"2307.11795","version":1},"attestation_state":"computed","paper":{"title":"Prompting Large Language Models with Speech Recognition Abilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"eess.AS","authors_text":"Christian Fuegen, Chunyang Wu, Egor Lakomkin, Jay Mahadeokar, Jinxi Guo, Junteng Jia, Ke Li, Mike Seltzer, Ozlem Kalinli, Wenhan Xiong, Yassir Fathullah, Yuan Shangguan","submitted_at":"2023-07-21T08:39:15Z","abstract_excerpt":"Large language models have proven themselves highly flexible, able to solve a wide range of generative tasks, such as abstractive summarization and open-ended question answering. In this paper we extend the capabilities of LLMs by directly attaching a small audio encoder allowing it to perform speech recognition. By directly prepending a sequence of audial embeddings to the text token embeddings, the LLM can be converted to an automatic speech recognition (ASR) system, and be used in the exact same manner as its textual counterpart. Experiments on Multilingual LibriSpeech (MLS) show that incor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.11795","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-07-21T08:39:15Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"a1342b05420d633a65a44d40652e2ad0f82d81b71960f944e17a1a5121557b61","abstract_canon_sha256":"0b1744add176fd5c606171b611845de6cdfa5b51b36ee115aead17fcac82ec40"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:33:34.897621Z","signature_b64":"kQGyrQWyFbllYBPN7LW4earS2j7Hbk+QCL3nrJmvhLeLlsacPTsSDZGG2H1MlSwe5tbHeEk7AgE7clKPVKA/CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d284f82fdb78bb34e7ba535a34dc7efbc3994707348fa17cd03dd8f6a6f8592","last_reissued_at":"2026-07-05T06:33:34.897128Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:33:34.897128Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prompting Large Language Models with Speech Recognition Abilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"eess.AS","authors_text":"Christian Fuegen, Chunyang Wu, Egor Lakomkin, Jay Mahadeokar, Jinxi Guo, Junteng Jia, Ke Li, Mike Seltzer, Ozlem Kalinli, Wenhan Xiong, Yassir Fathullah, Yuan Shangguan","submitted_at":"2023-07-21T08:39:15Z","abstract_excerpt":"Large language models have proven themselves highly flexible, able to solve a wide range of generative tasks, such as abstractive summarization and open-ended question answering. In this paper we extend the capabilities of LLMs by directly attaching a small audio encoder allowing it to perform speech recognition. By directly prepending a sequence of audial embeddings to the text token embeddings, the LLM can be converted to an automatic speech recognition (ASR) system, and be used in the exact same manner as its textual counterpart. Experiments on Multilingual LibriSpeech (MLS) show that incor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.11795","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.11795/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.11795","created_at":"2026-07-05T06:33:34.897187+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.11795v1","created_at":"2026-07-05T06:33:34.897187+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.11795","created_at":"2026-07-05T06:33:34.897187+00:00"},{"alias_kind":"pith_short_12","alias_value":"TUUE7AX5W6F3","created_at":"2026-07-05T06:33:34.897187+00:00"},{"alias_kind":"pith_short_16","alias_value":"TUUE7AX5W6F3GTT3","created_at":"2026-07-05T06:33:34.897187+00:00"},{"alias_kind":"pith_short_8","alias_value":"TUUE7AX5","created_at":"2026-07-05T06:33:34.897187+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10368","citing_title":"Speech Meets ELF: Audio Conditional Continuous-Target Diffusion for Speech Recognition and Translation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2310.13289","citing_title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","ref_index":107,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56","json":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56.json","graph_json":"https://pith.science/api/pith-number/TUUE7AX5W6F3GTT3UU22GTOH56/graph.json","events_json":"https://pith.science/api/pith-number/TUUE7AX5W6F3GTT3UU22GTOH56/events.json","paper":"https://pith.science/paper/TUUE7AX5"},"agent_actions":{"view_html":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56","download_json":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56.json","view_paper":"https://pith.science/paper/TUUE7AX5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.11795&json=true","fetch_graph":"https://pith.science/api/pith-number/TUUE7AX5W6F3GTT3UU22GTOH56/graph.json","fetch_events":"https://pith.science/api/pith-number/TUUE7AX5W6F3GTT3UU22GTOH56/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56/action/storage_attestation","attest_author":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56/action/author_attestation","sign_citation":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56/action/citation_signature","submit_replication":"https://pith.science/pith/TUUE7AX5W6F3GTT3UU22GTOH56/action/replication_record"}},"created_at":"2026-07-05T06:33:34.897187+00:00","updated_at":"2026-07-05T06:33:34.897187+00:00"}