{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IVUV5NFICQBIWY4AA2IX6AWVXV","short_pith_number":"pith:IVUV5NFI","schema_version":"1.0","canonical_sha256":"45695eb4a814028b638006917f02d5bd67553dbd808bf5f989f35cf858e2a7fa","source":{"kind":"arxiv","id":"2505.17496","version":1},"attestation_state":"computed","paper":{"title":"Analyzing Mitigation Strategies for Catastrophic Forgetting in End-to-End Training of Spoken Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chih-Kai Yang, Chi-Yuan Hsiao, Hung-yi Lee, Kai-Wei Chang, Ke-Han Lu, Wei-Chih Chen","submitted_at":"2025-05-23T05:50:14Z","abstract_excerpt":"End-to-end training of Spoken Language Models (SLMs) commonly involves adapting pre-trained text-based Large Language Models (LLMs) to the speech modality through multi-stage training on diverse tasks such as ASR, TTS and spoken question answering (SQA). Although this multi-stage continual learning equips LLMs with both speech understanding and generation capabilities, the substantial differences in task and data distributions across stages can lead to catastrophic forgetting, where previously acquired knowledge is lost. This paper investigates catastrophic forgetting and evaluates three mitig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17496","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T05:50:14Z","cross_cats_sorted":["cs.AI","cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"2ab233d82ea060ac1cf2f5d29bae6445d65883f1164e2a45e0b3ff0e7276037e","abstract_canon_sha256":"95f7dfd95363976de0252dc7b45f7a4b5f0943e793946d532e5e8408a818b705"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:27.049978Z","signature_b64":"7lYPWmx/5m8DrLxMU3mvdauAglSe4dS1QJ72awBeNm0qAW7e0h25Trwi/VRAXPbfYAIW7xZWibJFqhLnjzmrBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45695eb4a814028b638006917f02d5bd67553dbd808bf5f989f35cf858e2a7fa","last_reissued_at":"2026-07-05T11:08:27.049480Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:27.049480Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing Mitigation Strategies for Catastrophic Forgetting in End-to-End Training of Spoken Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chih-Kai Yang, Chi-Yuan Hsiao, Hung-yi Lee, Kai-Wei Chang, Ke-Han Lu, Wei-Chih Chen","submitted_at":"2025-05-23T05:50:14Z","abstract_excerpt":"End-to-end training of Spoken Language Models (SLMs) commonly involves adapting pre-trained text-based Large Language Models (LLMs) to the speech modality through multi-stage training on diverse tasks such as ASR, TTS and spoken question answering (SQA). Although this multi-stage continual learning equips LLMs with both speech understanding and generation capabilities, the substantial differences in task and data distributions across stages can lead to catastrophic forgetting, where previously acquired knowledge is lost. This paper investigates catastrophic forgetting and evaluates three mitig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17496","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17496/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17496","created_at":"2026-07-05T11:08:27.049542+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17496v1","created_at":"2026-07-05T11:08:27.049542+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17496","created_at":"2026-07-05T11:08:27.049542+00:00"},{"alias_kind":"pith_short_12","alias_value":"IVUV5NFICQBI","created_at":"2026-07-05T11:08:27.049542+00:00"},{"alias_kind":"pith_short_16","alias_value":"IVUV5NFICQBIWY4A","created_at":"2026-07-05T11:08:27.049542+00:00"},{"alias_kind":"pith_short_8","alias_value":"IVUV5NFI","created_at":"2026-07-05T11:08:27.049542+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24863","citing_title":"Rethinking Continual Learning for Speech and Audio: A Representation-Centric Taxonomy and Open Problems","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27393","citing_title":"MiniCPM-o 4.5: Towards Real-Time Full-Duplex Omni-Modal Interaction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05927","citing_title":"Minimizing Modality Gap from the Input Side: Your Speech LLM Can Be a Prosody-Aware Text LLM","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05927","citing_title":"Minimizing Modality Gap from the Input Side: Your Speech LLM Can Be a Prosody-Aware Text LLM","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV","json":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV.json","graph_json":"https://pith.science/api/pith-number/IVUV5NFICQBIWY4AA2IX6AWVXV/graph.json","events_json":"https://pith.science/api/pith-number/IVUV5NFICQBIWY4AA2IX6AWVXV/events.json","paper":"https://pith.science/paper/IVUV5NFI"},"agent_actions":{"view_html":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV","download_json":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV.json","view_paper":"https://pith.science/paper/IVUV5NFI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17496&json=true","fetch_graph":"https://pith.science/api/pith-number/IVUV5NFICQBIWY4AA2IX6AWVXV/graph.json","fetch_events":"https://pith.science/api/pith-number/IVUV5NFICQBIWY4AA2IX6AWVXV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV/action/storage_attestation","attest_author":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV/action/author_attestation","sign_citation":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV/action/citation_signature","submit_replication":"https://pith.science/pith/IVUV5NFICQBIWY4AA2IX6AWVXV/action/replication_record"}},"created_at":"2026-07-05T11:08:27.049542+00:00","updated_at":"2026-07-05T11:08:27.049542+00:00"}