{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RQVLMQOOMVIFSB3JTJJOYIH7SZ","short_pith_number":"pith:RQVLMQOO","schema_version":"1.0","canonical_sha256":"8c2ab641ce65505907699a52ec20ff965aecf3d452bd03ded17739b2ad1de5ba","source":{"kind":"arxiv","id":"2309.13963","version":2},"attestation_state":"computed","paper":{"title":"Connecting Speech Encoder and Large Language Model for ASR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Changli Tang, Chao Zhang, Guangzhi Sun, Lu Lu, Tian Tan, Wei Li, Wenyi Yu, Xianzhao Chen, Zejun Ma","submitted_at":"2023-09-25T08:57:07Z","abstract_excerpt":"The impressive capability and versatility of large language models (LLMs) have aroused increasing attention in automatic speech recognition (ASR), with several pioneering studies attempting to build integrated ASR models by connecting a speech encoder with an LLM. This paper presents a comparative study of three commonly used structures as connectors, including fully connected layers, multi-head cross-attention, and Q-Former. Speech encoders from the Whisper model series as well as LLMs from the Vicuna model series with different model sizes were studied. Experiments were performed on the comm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.13963","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-09-25T08:57:07Z","cross_cats_sorted":["cs.CL","cs.SD"],"title_canon_sha256":"248bb5217f636418fab20b77591f489cca4dac1347f884a258004237aaaab1e6","abstract_canon_sha256":"46f0031864208963eab321ca7aa150c3885fd448122992401abc9116e18051ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:54:24.048105Z","signature_b64":"9QGys32P04x0mqmIZi5GNbjRcif7Fd6jFIJ+Vk+HLtd+GuDe9aLzc3JYYkwYtULd1eTByj0UWHFf0MgSdEEXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c2ab641ce65505907699a52ec20ff965aecf3d452bd03ded17739b2ad1de5ba","last_reissued_at":"2026-07-05T06:54:24.047614Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:54:24.047614Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Connecting Speech Encoder and Large Language Model for ASR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Changli Tang, Chao Zhang, Guangzhi Sun, Lu Lu, Tian Tan, Wei Li, Wenyi Yu, Xianzhao Chen, Zejun Ma","submitted_at":"2023-09-25T08:57:07Z","abstract_excerpt":"The impressive capability and versatility of large language models (LLMs) have aroused increasing attention in automatic speech recognition (ASR), with several pioneering studies attempting to build integrated ASR models by connecting a speech encoder with an LLM. This paper presents a comparative study of three commonly used structures as connectors, including fully connected layers, multi-head cross-attention, and Q-Former. Speech encoders from the Whisper model series as well as LLMs from the Vicuna model series with different model sizes were studied. Experiments were performed on the comm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.13963","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.13963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.13963","created_at":"2026-07-05T06:54:24.047686+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.13963v2","created_at":"2026-07-05T06:54:24.047686+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.13963","created_at":"2026-07-05T06:54:24.047686+00:00"},{"alias_kind":"pith_short_12","alias_value":"RQVLMQOOMVIF","created_at":"2026-07-05T06:54:24.047686+00:00"},{"alias_kind":"pith_short_16","alias_value":"RQVLMQOOMVIFSB3J","created_at":"2026-07-05T06:54:24.047686+00:00"},{"alias_kind":"pith_short_8","alias_value":"RQVLMQOO","created_at":"2026-07-05T06:54:24.047686+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10368","citing_title":"Speech Meets ELF: Audio Conditional Continuous-Target Diffusion for Speech Recognition and Translation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2310.13289","citing_title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15045","citing_title":"LLMs and Speech: Integration vs. Combination","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09332","citing_title":"Phonemes vs. Projectors: An Investigation of Speech-Language Interfaces for LLM-based ASR","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ","json":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ.json","graph_json":"https://pith.science/api/pith-number/RQVLMQOOMVIFSB3JTJJOYIH7SZ/graph.json","events_json":"https://pith.science/api/pith-number/RQVLMQOOMVIFSB3JTJJOYIH7SZ/events.json","paper":"https://pith.science/paper/RQVLMQOO"},"agent_actions":{"view_html":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ","download_json":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ.json","view_paper":"https://pith.science/paper/RQVLMQOO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.13963&json=true","fetch_graph":"https://pith.science/api/pith-number/RQVLMQOOMVIFSB3JTJJOYIH7SZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RQVLMQOOMVIFSB3JTJJOYIH7SZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ/action/storage_attestation","attest_author":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ/action/author_attestation","sign_citation":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ/action/citation_signature","submit_replication":"https://pith.science/pith/RQVLMQOOMVIFSB3JTJJOYIH7SZ/action/replication_record"}},"created_at":"2026-07-05T06:54:24.047686+00:00","updated_at":"2026-07-05T06:54:24.047686+00:00"}