{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P5LFSZZ37Y26XAW7UPNETHDQXR","short_pith_number":"pith:P5LFSZZ3","schema_version":"1.0","canonical_sha256":"7f5659673bfe35eb82dfa3da499c70bc79e53ce93612836c1b1827643217500a","source":{"kind":"arxiv","id":"2407.08551","version":2},"attestation_state":"computed","paper":{"title":"Autoregressive Speech Synthesis without Vector Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bing Han, Furu Wei, Helen Meng, Jinyu Li, Lingwei Meng, Long Zhou, Sanyuan Chen, Sheng Zhao, Shujie Hu, Shujie Liu, Xixin Wu, Yanqing Liu","submitted_at":"2024-07-11T14:36:53Z","abstract_excerpt":"We present MELLE, a novel continuous-valued token based language modeling approach for text-to-speech synthesis (TTS). MELLE autoregressively generates continuous mel-spectrogram frames directly from text condition, bypassing the need for vector quantization, which is typically designed for audio compression and sacrifices fidelity compared to continuous representations. Specifically, (i) instead of cross-entropy loss, we apply regression loss with a proposed spectrogram flux loss function to model the probability distribution of the continuous-valued tokens; (ii) we have incorporated variatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08551","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-11T14:36:53Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"a17ac2791bdffd09859153d34a1a88c28886d7951ef9f289837f7ec9b9cf78ca","abstract_canon_sha256":"c03dea6507bf8ac024bc38383807f3ff746d098bf1fb5c74e926097cb8ee416b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:59.871021Z","signature_b64":"SCzZFQqePURVY7X80dYX96lgC1Z//6wuX92FTKDEictVdTgL4/vyeApxPaxYvtKuHLS3bmbztjE+LEXD2I8DBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f5659673bfe35eb82dfa3da499c70bc79e53ce93612836c1b1827643217500a","last_reissued_at":"2026-07-05T11:09:59.870501Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:59.870501Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Autoregressive Speech Synthesis without Vector Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bing Han, Furu Wei, Helen Meng, Jinyu Li, Lingwei Meng, Long Zhou, Sanyuan Chen, Sheng Zhao, Shujie Hu, Shujie Liu, Xixin Wu, Yanqing Liu","submitted_at":"2024-07-11T14:36:53Z","abstract_excerpt":"We present MELLE, a novel continuous-valued token based language modeling approach for text-to-speech synthesis (TTS). MELLE autoregressively generates continuous mel-spectrogram frames directly from text condition, bypassing the need for vector quantization, which is typically designed for audio compression and sacrifices fidelity compared to continuous representations. Specifically, (i) instead of cross-entropy loss, we apply regression loss with a proposed spectrogram flux loss function to model the probability distribution of the continuous-valued tokens; (ii) we have incorporated variatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08551","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08551/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08551","created_at":"2026-07-05T11:09:59.870565+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08551v2","created_at":"2026-07-05T11:09:59.870565+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08551","created_at":"2026-07-05T11:09:59.870565+00:00"},{"alias_kind":"pith_short_12","alias_value":"P5LFSZZ37Y26","created_at":"2026-07-05T11:09:59.870565+00:00"},{"alias_kind":"pith_short_16","alias_value":"P5LFSZZ37Y26XAW7","created_at":"2026-07-05T11:09:59.870565+00:00"},{"alias_kind":"pith_short_8","alias_value":"P5LFSZZ3","created_at":"2026-07-05T11:09:59.870565+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27978","citing_title":"Parallel Rollout Approximation for Pixel-Space Autoregressive Image Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06885","citing_title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17589","citing_title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10117","citing_title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06327","citing_title":"A Novel Automatic Framework for Speaker Drift Detection in Synthesized Speech","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR","json":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR.json","graph_json":"https://pith.science/api/pith-number/P5LFSZZ37Y26XAW7UPNETHDQXR/graph.json","events_json":"https://pith.science/api/pith-number/P5LFSZZ37Y26XAW7UPNETHDQXR/events.json","paper":"https://pith.science/paper/P5LFSZZ3"},"agent_actions":{"view_html":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR","download_json":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR.json","view_paper":"https://pith.science/paper/P5LFSZZ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08551&json=true","fetch_graph":"https://pith.science/api/pith-number/P5LFSZZ37Y26XAW7UPNETHDQXR/graph.json","fetch_events":"https://pith.science/api/pith-number/P5LFSZZ37Y26XAW7UPNETHDQXR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR/action/storage_attestation","attest_author":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR/action/author_attestation","sign_citation":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR/action/citation_signature","submit_replication":"https://pith.science/pith/P5LFSZZ37Y26XAW7UPNETHDQXR/action/replication_record"}},"created_at":"2026-07-05T11:09:59.870565+00:00","updated_at":"2026-07-05T11:09:59.870565+00:00"}