{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7GYXLXVCKMHTI77UVSPVOFS5UD","short_pith_number":"pith:7GYXLXVC","schema_version":"1.0","canonical_sha256":"f9b175dea2530f347ff4ac9f57165da0c6926b79cd22002d46bfb655d4fef580","source":{"kind":"arxiv","id":"2406.07855","version":1},"attestation_state":"computed","paper":{"title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bing Han, Furu Wei, Jinyu Li, Lingwei Meng, Long Zhou, Sanyuan Chen, Sheng Zhao, Shujie Liu, Yanming Qian, Yanqing Liu","submitted_at":"2024-06-12T04:09:44Z","abstract_excerpt":"With the help of discrete neural audio codecs, large language models (LLM) have increasingly been recognized as a promising methodology for zero-shot Text-to-Speech (TTS) synthesis. However, sampling based decoding strategies bring astonishing diversity to generation, but also pose robustness issues such as typos, omissions and repetition. In addition, the high sampling rate of audio also brings huge computational overhead to the inference process of autoregression. To address these issues, we propose VALL-E R, a robust and efficient zero-shot TTS system, building upon the foundation of VALL-E"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07855","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T04:09:44Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"5c9b682b567e23af8700b63aac4edc9f7625671f0824eb66ec789e50d4c2fd5d","abstract_canon_sha256":"e710f8bd7d605dde2f1c9f63abbc15c53d9719e0b10d7ad028f24f810caa6ae0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:39.542221Z","signature_b64":"lytlT3uT+WrDGye0I4SttuFKK0GNsUmmaE/uY9BazCJKDEqfBxFKar7ilxmiLbuErn1J1LF011TFTLmVi3U+DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9b175dea2530f347ff4ac9f57165da0c6926b79cd22002d46bfb655d4fef580","last_reissued_at":"2026-07-05T08:30:39.541717Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:39.541717Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bing Han, Furu Wei, Jinyu Li, Lingwei Meng, Long Zhou, Sanyuan Chen, Sheng Zhao, Shujie Liu, Yanming Qian, Yanqing Liu","submitted_at":"2024-06-12T04:09:44Z","abstract_excerpt":"With the help of discrete neural audio codecs, large language models (LLM) have increasingly been recognized as a promising methodology for zero-shot Text-to-Speech (TTS) synthesis. However, sampling based decoding strategies bring astonishing diversity to generation, but also pose robustness issues such as typos, omissions and repetition. In addition, the high sampling rate of audio also brings huge computational overhead to the inference process of autoregression. To address these issues, we propose VALL-E R, a robust and efficient zero-shot TTS system, building upon the foundation of VALL-E"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07855","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07855/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07855","created_at":"2026-07-05T08:30:39.541776+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07855v1","created_at":"2026-07-05T08:30:39.541776+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07855","created_at":"2026-07-05T08:30:39.541776+00:00"},{"alias_kind":"pith_short_12","alias_value":"7GYXLXVCKMHT","created_at":"2026-07-05T08:30:39.541776+00:00"},{"alias_kind":"pith_short_16","alias_value":"7GYXLXVCKMHTI77U","created_at":"2026-07-05T08:30:39.541776+00:00"},{"alias_kind":"pith_short_8","alias_value":"7GYXLXVC","created_at":"2026-07-05T08:30:39.541776+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18485","citing_title":"MagpieTTS-LF: Inference-Time Long-Form Speech Generation Without Training on Long-Form data","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03455","citing_title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01677","citing_title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16964","citing_title":"SemaVoice: Semantic-Aware Continuous Autoregressive Speech Synthesis","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19883","citing_title":"CoMelSinger: Discrete Token-Based Zero-Shot Singing Synthesis With Structured Melody Control and Guidance","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06885","citing_title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17589","citing_title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05373","citing_title":"Hierarchical Decoding for Discrete Speech Synthesis with Multi-Resolution Spoof Detection","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14267","citing_title":"DiFlowDubber: Discrete Flow Matching for Automated Video Dubbing via Cross-Modal Alignment and Synchronization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10117","citing_title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD","json":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD.json","graph_json":"https://pith.science/api/pith-number/7GYXLXVCKMHTI77UVSPVOFS5UD/graph.json","events_json":"https://pith.science/api/pith-number/7GYXLXVCKMHTI77UVSPVOFS5UD/events.json","paper":"https://pith.science/paper/7GYXLXVC"},"agent_actions":{"view_html":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD","download_json":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD.json","view_paper":"https://pith.science/paper/7GYXLXVC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07855&json=true","fetch_graph":"https://pith.science/api/pith-number/7GYXLXVCKMHTI77UVSPVOFS5UD/graph.json","fetch_events":"https://pith.science/api/pith-number/7GYXLXVCKMHTI77UVSPVOFS5UD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD/action/storage_attestation","attest_author":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD/action/author_attestation","sign_citation":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD/action/citation_signature","submit_replication":"https://pith.science/pith/7GYXLXVCKMHTI77UVSPVOFS5UD/action/replication_record"}},"created_at":"2026-07-05T08:30:39.541776+00:00","updated_at":"2026-07-05T08:30:39.541776+00:00"}