{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DFRYGFCH4YSX4KDA6FEMN32G5A","short_pith_number":"pith:DFRYGFCH","schema_version":"1.0","canonical_sha256":"1963831447e6257e2860f148c6ef46e8355dcfb61b62e4816b16a5b039d98821","source":{"kind":"arxiv","id":"2302.08917","version":1},"attestation_state":"computed","paper":{"title":"Massively Multilingual Shallow Fusion with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew M. Dai, Bo Li, Ke Hu, Nan Du, Rodrigo Cabrera, Tara N. Sainath, Trevor Strohman, Yanping Huang, Yu Zhang, Zhifeng Chen","submitted_at":"2023-02-17T14:46:38Z","abstract_excerpt":"While large language models (LLM) have made impressive progress in natural language processing, it remains unclear how to utilize them in improving automatic speech recognition (ASR). In this work, we propose to train a single multilingual language model (LM) for shallow fusion in multiple languages. We push the limits of the multilingual LM to cover up to 84 languages by scaling up using a mixture-of-experts LLM, i.e., generalist language model (GLaM). When the number of experts increases, GLaM dynamically selects only two at each decoding step to keep the inference computation roughly consta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08917","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-17T14:46:38Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1f11e2c30a67b58edd5dbf20fca834071056666b489e6362b3e7ae700d476edc","abstract_canon_sha256":"ac2b7735f2f4e9996b775cae3c1c17b49e89b0931f3dbc6ce4d9e7655eca9e7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:42:56.553614Z","signature_b64":"/3106Fo78NB6Zl0mKGqz9UWIjardobw3vMCbuRKsdr0b8ZG0jwwphxkECWiPuYlIwc29SQCLUXOOzvrXqgX2AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1963831447e6257e2860f148c6ef46e8355dcfb61b62e4816b16a5b039d98821","last_reissued_at":"2026-07-05T05:42:56.553180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:42:56.553180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Massively Multilingual Shallow Fusion with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew M. Dai, Bo Li, Ke Hu, Nan Du, Rodrigo Cabrera, Tara N. Sainath, Trevor Strohman, Yanping Huang, Yu Zhang, Zhifeng Chen","submitted_at":"2023-02-17T14:46:38Z","abstract_excerpt":"While large language models (LLM) have made impressive progress in natural language processing, it remains unclear how to utilize them in improving automatic speech recognition (ASR). In this work, we propose to train a single multilingual language model (LM) for shallow fusion in multiple languages. We push the limits of the multilingual LM to cover up to 84 languages by scaling up using a mixture-of-experts LLM, i.e., generalist language model (GLaM). When the number of experts increases, GLaM dynamically selects only two at each decoding step to keep the inference computation roughly consta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08917","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08917/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08917","created_at":"2026-07-05T05:42:56.553243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08917v1","created_at":"2026-07-05T05:42:56.553243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08917","created_at":"2026-07-05T05:42:56.553243+00:00"},{"alias_kind":"pith_short_12","alias_value":"DFRYGFCH4YSX","created_at":"2026-07-05T05:42:56.553243+00:00"},{"alias_kind":"pith_short_16","alias_value":"DFRYGFCH4YSX4KDA","created_at":"2026-07-05T05:42:56.553243+00:00"},{"alias_kind":"pith_short_8","alias_value":"DFRYGFCH","created_at":"2026-07-05T05:42:56.553243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A","json":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A.json","graph_json":"https://pith.science/api/pith-number/DFRYGFCH4YSX4KDA6FEMN32G5A/graph.json","events_json":"https://pith.science/api/pith-number/DFRYGFCH4YSX4KDA6FEMN32G5A/events.json","paper":"https://pith.science/paper/DFRYGFCH"},"agent_actions":{"view_html":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A","download_json":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A.json","view_paper":"https://pith.science/paper/DFRYGFCH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08917&json=true","fetch_graph":"https://pith.science/api/pith-number/DFRYGFCH4YSX4KDA6FEMN32G5A/graph.json","fetch_events":"https://pith.science/api/pith-number/DFRYGFCH4YSX4KDA6FEMN32G5A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A/action/storage_attestation","attest_author":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A/action/author_attestation","sign_citation":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A/action/citation_signature","submit_replication":"https://pith.science/pith/DFRYGFCH4YSX4KDA6FEMN32G5A/action/replication_record"}},"created_at":"2026-07-05T05:42:56.553243+00:00","updated_at":"2026-07-05T05:42:56.553243+00:00"}