{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M3ESM2RLMWSWR45HCQJAYRWYFX","short_pith_number":"pith:M3ESM2RL","schema_version":"1.0","canonical_sha256":"66c9266a2b65a568f3a714120c46d82dd0fa92003220de657bea1097988b5d59","source":{"kind":"arxiv","id":"2508.04096","version":1},"attestation_state":"computed","paper":{"title":"Efficient Scaling for LLM-based ASR","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Bingshen Mu, Dong Yu, Kun Wei, Lei Xie, Yiwen Shao","submitted_at":"2025-08-06T05:28:18Z","abstract_excerpt":"Large language model (LLM)-based automatic speech recognition (ASR) achieves strong performance but often incurs high computational costs. This work investigates how to obtain the best LLM-ASR performance efficiently. Through comprehensive and controlled experiments, we find that pretraining the speech encoder before integrating it with the LLM leads to significantly better scaling efficiency than the standard practice of joint post-training of LLM-ASR. Based on this insight, we propose a new multi-stage LLM-ASR training strategy, EFIN: Encoder First Integration. Among all training strategies "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.04096","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.SD","submitted_at":"2025-08-06T05:28:18Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"173fe38c0966606d83a3fb7b589caece25b4e4eb991acb90cd9f31293eaf5595","abstract_canon_sha256":"d9ea12ba6af3e621a1d1ccf5076e4d45bc3ffe69c0713efdce29cc77c7cc1ddc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:20.532049Z","signature_b64":"c0Ydc88zM8t/iVaPzvTufl0xwJnAHuoM4GquYK3TiV8wiJHuwIOtZln1+jiUq6dFpij5LyAXZjQUc1zpqHeLBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66c9266a2b65a568f3a714120c46d82dd0fa92003220de657bea1097988b5d59","last_reissued_at":"2026-07-05T11:49:20.531534Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:20.531534Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Scaling for LLM-based ASR","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Bingshen Mu, Dong Yu, Kun Wei, Lei Xie, Yiwen Shao","submitted_at":"2025-08-06T05:28:18Z","abstract_excerpt":"Large language model (LLM)-based automatic speech recognition (ASR) achieves strong performance but often incurs high computational costs. This work investigates how to obtain the best LLM-ASR performance efficiently. Through comprehensive and controlled experiments, we find that pretraining the speech encoder before integrating it with the LLM leads to significantly better scaling efficiency than the standard practice of joint post-training of LLM-ASR. Based on this insight, we propose a new multi-stage LLM-ASR training strategy, EFIN: Encoder First Integration. Among all training strategies "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.04096","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.04096/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.04096","created_at":"2026-07-05T11:49:20.531592+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.04096v1","created_at":"2026-07-05T11:49:20.531592+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.04096","created_at":"2026-07-05T11:49:20.531592+00:00"},{"alias_kind":"pith_short_12","alias_value":"M3ESM2RLMWSW","created_at":"2026-07-05T11:49:20.531592+00:00"},{"alias_kind":"pith_short_16","alias_value":"M3ESM2RLMWSWR45H","created_at":"2026-07-05T11:49:20.531592+00:00"},{"alias_kind":"pith_short_8","alias_value":"M3ESM2RL","created_at":"2026-07-05T11:49:20.531592+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01733","citing_title":"Rethinking Speech-LLM Integration for ASR: Effective Joint Speech-Text Training by Interleaving","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14804","citing_title":"Towards Building Speech Large Language Models for Multitask Understanding in Low-Resource Languages","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX","json":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX.json","graph_json":"https://pith.science/api/pith-number/M3ESM2RLMWSWR45HCQJAYRWYFX/graph.json","events_json":"https://pith.science/api/pith-number/M3ESM2RLMWSWR45HCQJAYRWYFX/events.json","paper":"https://pith.science/paper/M3ESM2RL"},"agent_actions":{"view_html":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX","download_json":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX.json","view_paper":"https://pith.science/paper/M3ESM2RL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.04096&json=true","fetch_graph":"https://pith.science/api/pith-number/M3ESM2RLMWSWR45HCQJAYRWYFX/graph.json","fetch_events":"https://pith.science/api/pith-number/M3ESM2RLMWSWR45HCQJAYRWYFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX/action/storage_attestation","attest_author":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX/action/author_attestation","sign_citation":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX/action/citation_signature","submit_replication":"https://pith.science/pith/M3ESM2RLMWSWR45HCQJAYRWYFX/action/replication_record"}},"created_at":"2026-07-05T11:49:20.531592+00:00","updated_at":"2026-07-05T11:49:20.531592+00:00"}