{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IZURCTWC2VRFVLY6CURCUOIR5V","short_pith_number":"pith:IZURCTWC","schema_version":"1.0","canonical_sha256":"4669114ec2d5625aaf1e15222a3911ed510354f6b030b2da455374bda674e54b","source":{"kind":"arxiv","id":"2502.04128","version":2},"attestation_state":"computed","paper":{"title":"Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.MM","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chi-Min Chan, Haohe Liu, Hongzhan Lin, Jiahe Lei, Jianyi Chen, Lei Xie, Liumeng Xue, Qiuqiang Kong, Wei Xue, Xinfa Zhu, Xingjian Du, Xinsheng Wang, Xu Tan, Yike Guo, Yi Peng, Yizhu Jin, Yunlin Chen, Zhen Ye, Zheqi Dai, Zhifei Li","submitted_at":"2025-02-06T15:04:00Z","abstract_excerpt":"Recent advances in text-based large language models (LLMs), particularly in the GPT series and the o1 model, have demonstrated the effectiveness of scaling both training-time and inference-time compute. However, current state-of-the-art TTS systems leveraging LLMs are often multi-stage, requiring separate models (e.g., diffusion models after LLM), complicating the decision of whether to scale a particular model during training or testing. This work makes the following contributions: First, we explore the scaling of train-time and inference-time compute for speech synthesis. Second, we propose "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04128","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"eess.AS","submitted_at":"2025-02-06T15:04:00Z","cross_cats_sorted":["cs.AI","cs.CL","cs.MM","cs.SD"],"title_canon_sha256":"321672531db2d8aa3f98788954de7c3d2f2d9655440203a04ffd99909c0b3d9c","abstract_canon_sha256":"f91318cc083f36c97b49659bad97ac6d65b6234f25f0ec937bcb4b3cc2b1d73f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:13.177294Z","signature_b64":"HNvODzO0ET31GGogqphTP8cl8T8avPLNJcHA4Y6c38iN4uEbgo7AcTfGLVnvZsCiwBRUHK2Mwr6lCctboOHzAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4669114ec2d5625aaf1e15222a3911ed510354f6b030b2da455374bda674e54b","last_reissued_at":"2026-07-05T10:18:13.176802Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:13.176802Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.MM","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chi-Min Chan, Haohe Liu, Hongzhan Lin, Jiahe Lei, Jianyi Chen, Lei Xie, Liumeng Xue, Qiuqiang Kong, Wei Xue, Xinfa Zhu, Xingjian Du, Xinsheng Wang, Xu Tan, Yike Guo, Yi Peng, Yizhu Jin, Yunlin Chen, Zhen Ye, Zheqi Dai, Zhifei Li","submitted_at":"2025-02-06T15:04:00Z","abstract_excerpt":"Recent advances in text-based large language models (LLMs), particularly in the GPT series and the o1 model, have demonstrated the effectiveness of scaling both training-time and inference-time compute. However, current state-of-the-art TTS systems leveraging LLMs are often multi-stage, requiring separate models (e.g., diffusion models after LLM), complicating the decision of whether to scale a particular model during training or testing. This work makes the following contributions: First, we explore the scaling of train-time and inference-time compute for speech synthesis. Second, we propose "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04128","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04128/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04128","created_at":"2026-07-05T10:18:13.176863+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04128v2","created_at":"2026-07-05T10:18:13.176863+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04128","created_at":"2026-07-05T10:18:13.176863+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZURCTWC2VRF","created_at":"2026-07-05T10:18:13.176863+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZURCTWC2VRFVLY6","created_at":"2026-07-05T10:18:13.176863+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZURCTWC","created_at":"2026-07-05T10:18:13.176863+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":35,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":184,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23190","citing_title":"FlowTTS-GRPO: Online Reinforcement Learning with Multi-Objective Reward Optimization for Flow-Matching Based Text-to-Speech","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21888","citing_title":"ProsoCodec: Prosody-Oriented Speech Codec for Voice Conversion","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18072","citing_title":"One-Step Token-to-Waveform Generation with MeanFlow in Latent Space","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18323","citing_title":"Reliable Neural-Codec Text-to-Speech by ASR Self-Verification and Distillation: Near-Zero Catastrophic Failures Across Models and Codecs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12940","citing_title":"Self-Guidance: Enhancing Neural Codecs via Decoder Manifold Alignment","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09141","citing_title":"FlashTTS: Fast Streaming TTS with MTP Acceleration and X-pred Mean Flow Distillation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09019","citing_title":"TLDR: Compressing Audio Tokens for Efficient Autoregressive Text-to-Speech","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07080","citing_title":"dots.tts Technical Report","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06928","citing_title":"VoxCPM2 Technical Report","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03455","citing_title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":207,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29480","citing_title":"DTM-Codec: Dynamic Token Masking for VFR Speech Coding with Efficient Boundary Selection","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02739","citing_title":"EntangleCodec: A Unified Discrete Audio Tokenizer via Semantic-Acoustic Entanglement","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04418","citing_title":"CleanCodec: Efficient and Robust Speech Tokenization via Perceptually Guided Encoding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01537","citing_title":"Two-Dimensional Quantization for Geometry-Aware Audio Coding","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20830","citing_title":"Raon-OpenTTS: Open Models and Data for Robust Text-to-Speech","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19541","citing_title":"Optimising Neural Speech Codecs for 300bps Communication using Reinforcement Learning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16964","citing_title":"SemaVoice: Semantic-Aware Continuous Autoregressive Speech Synthesis","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15621","citing_title":"Qwen3-TTS Technical Report","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22143","citing_title":"JUST-DUB-IT: Video Dubbing via Joint Audio-Visual Diffusion","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05373","citing_title":"Hierarchical Decoding for Discrete Speech Synthesis with Multi-Resolution Spoof Detection","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14555","citing_title":"Break-the-Beat! Controllable MIDI-to-Drum Audio Synthesis","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00688","citing_title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V","json":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V.json","graph_json":"https://pith.science/api/pith-number/IZURCTWC2VRFVLY6CURCUOIR5V/graph.json","events_json":"https://pith.science/api/pith-number/IZURCTWC2VRFVLY6CURCUOIR5V/events.json","paper":"https://pith.science/paper/IZURCTWC"},"agent_actions":{"view_html":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V","download_json":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V.json","view_paper":"https://pith.science/paper/IZURCTWC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04128&json=true","fetch_graph":"https://pith.science/api/pith-number/IZURCTWC2VRFVLY6CURCUOIR5V/graph.json","fetch_events":"https://pith.science/api/pith-number/IZURCTWC2VRFVLY6CURCUOIR5V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V/action/storage_attestation","attest_author":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V/action/author_attestation","sign_citation":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V/action/citation_signature","submit_replication":"https://pith.science/pith/IZURCTWC2VRFVLY6CURCUOIR5V/action/replication_record"}},"created_at":"2026-07-05T10:18:13.176863+00:00","updated_at":"2026-07-05T10:18:13.176863+00:00"}