{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2M6WQJNBZQLJBPGY4KFR4LMA2G","short_pith_number":"pith:2M6WQJNB","schema_version":"1.0","canonical_sha256":"d33d6825a1cc1690bcd8e28b1e2d80d1a778810d97771ae5d99adc953dad0aee","source":{"kind":"arxiv","id":"2412.15649","version":1},"attestation_state":"computed","paper":{"title":"SLAM-Omni: Timbre-Controllable Voice Interaction System with Single-Stage Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Jinyu Li, Kai Yu, Ruiqi Yan, Ruiyang Xu, Shujie Liu, Wenxi Chen, Xie Chen, Xiquan Li, Yan Lu, Yanqiao Zhu, Yifan Yang, Yuxuan Hu, Yuzhe Liang, Zhanxun Liu, Zhikang Niu, Ziyang Ma","submitted_at":"2024-12-20T08:05:55Z","abstract_excerpt":"Recent advancements highlight the potential of end-to-end real-time spoken dialogue systems, showcasing their low latency and high quality. In this paper, we introduce SLAM-Omni, a timbre-controllable, end-to-end voice interaction system with single-stage training. SLAM-Omni achieves zero-shot timbre control by modeling spoken language with semantic tokens and decoupling speaker information to a vocoder. By predicting grouped speech semantic tokens at each step, our method significantly reduces the sequence length of audio tokens, accelerating both training and inference. Additionally, we prop"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15649","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-12-20T08:05:55Z","cross_cats_sorted":[],"title_canon_sha256":"33eafac155c01e7f8d7610a2d6540060e3174021a6f5f0f2b0bc0969980ab7b7","abstract_canon_sha256":"33527fbbec7a1e952540f8887e5724682fad2aa7692cd7720c177bf12451308f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:27.671171Z","signature_b64":"l+WvR3UAhIrqiKvYgyjpXqAooU62DoLjn12YhBXXVcIWe78pmJnJzzx9oMNpOO8MT4ZDNBo1LpJZBUIDiSD/CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d33d6825a1cc1690bcd8e28b1e2d80d1a778810d97771ae5d99adc953dad0aee","last_reissued_at":"2026-07-05T09:52:27.670589Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:27.670589Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SLAM-Omni: Timbre-Controllable Voice Interaction System with Single-Stage Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Jinyu Li, Kai Yu, Ruiqi Yan, Ruiyang Xu, Shujie Liu, Wenxi Chen, Xie Chen, Xiquan Li, Yan Lu, Yanqiao Zhu, Yifan Yang, Yuxuan Hu, Yuzhe Liang, Zhanxun Liu, Zhikang Niu, Ziyang Ma","submitted_at":"2024-12-20T08:05:55Z","abstract_excerpt":"Recent advancements highlight the potential of end-to-end real-time spoken dialogue systems, showcasing their low latency and high quality. In this paper, we introduce SLAM-Omni, a timbre-controllable, end-to-end voice interaction system with single-stage training. SLAM-Omni achieves zero-shot timbre control by modeling spoken language with semantic tokens and decoupling speaker information to a vocoder. By predicting grouped speech semantic tokens at each step, our method significantly reduces the sequence length of audio tokens, accelerating both training and inference. Additionally, we prop"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15649","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15649/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15649","created_at":"2026-07-05T09:52:27.670648+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15649v1","created_at":"2026-07-05T09:52:27.670648+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15649","created_at":"2026-07-05T09:52:27.670648+00:00"},{"alias_kind":"pith_short_12","alias_value":"2M6WQJNBZQLJ","created_at":"2026-07-05T09:52:27.670648+00:00"},{"alias_kind":"pith_short_16","alias_value":"2M6WQJNBZQLJBPGY","created_at":"2026-07-05T09:52:27.670648+00:00"},{"alias_kind":"pith_short_8","alias_value":"2M6WQJNB","created_at":"2026-07-05T09:52:27.670648+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12199","citing_title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2504.08528","citing_title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06765","citing_title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13804","citing_title":"Character Beyond Speech: Leveraging Role-Playing Evaluation in Audio Large Language Models via Reinforcement Learning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G","json":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G.json","graph_json":"https://pith.science/api/pith-number/2M6WQJNBZQLJBPGY4KFR4LMA2G/graph.json","events_json":"https://pith.science/api/pith-number/2M6WQJNBZQLJBPGY4KFR4LMA2G/events.json","paper":"https://pith.science/paper/2M6WQJNB"},"agent_actions":{"view_html":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G","download_json":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G.json","view_paper":"https://pith.science/paper/2M6WQJNB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15649&json=true","fetch_graph":"https://pith.science/api/pith-number/2M6WQJNBZQLJBPGY4KFR4LMA2G/graph.json","fetch_events":"https://pith.science/api/pith-number/2M6WQJNBZQLJBPGY4KFR4LMA2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G/action/storage_attestation","attest_author":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G/action/author_attestation","sign_citation":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G/action/citation_signature","submit_replication":"https://pith.science/pith/2M6WQJNBZQLJBPGY4KFR4LMA2G/action/replication_record"}},"created_at":"2026-07-05T09:52:27.670648+00:00","updated_at":"2026-07-05T09:52:27.670648+00:00"}