{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:TWTD3DMTKGQISEUWJ5GOJWNCPA","short_pith_number":"pith:TWTD3DMT","schema_version":"1.0","canonical_sha256":"9da63d8d9351a08912964f4ce4d9a2783925766df730eb4121da8518cea65a3e","source":{"kind":"arxiv","id":"2005.11129","version":2},"attestation_state":"computed","paper":{"title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Jaehyeon Kim, Jungil Kong, Sungroh Yoon, Sungwon Kim","submitted_at":"2020-05-22T12:06:46Z","abstract_excerpt":"Recently, text-to-speech (TTS) models such as FastSpeech and ParaNet have been proposed to generate mel-spectrograms from text in parallel. Despite the advantage, the parallel TTS models cannot be trained without guidance from autoregressive TTS models as their external aligners. In this work, we propose Glow-TTS, a flow-based generative model for parallel TTS that does not require any external aligner. By combining the properties of flows and dynamic programming, the proposed model searches for the most probable monotonic alignment between text and the latent representation of speech on its o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.11129","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-05-22T12:06:46Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"e171a809c8301f637afd40b8cfb6d1fbb76091f25ba4b4c59171d82664735bf9","abstract_canon_sha256":"807a89609a433e8b6fd1f1dfb9d9ebdc7a522493cee914bf9b1c8ce138c434b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:27.073873Z","signature_b64":"p6H74iun7UatR7qtrzKJpst0vxca0ue9eLoFTn1k+3NlECVOR5Hf3Aiw4aQxysIp2YLj9X7TXvAzG4SPFRUmDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9da63d8d9351a08912964f4ce4d9a2783925766df730eb4121da8518cea65a3e","last_reissued_at":"2026-07-05T01:45:27.073385Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:27.073385Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Jaehyeon Kim, Jungil Kong, Sungroh Yoon, Sungwon Kim","submitted_at":"2020-05-22T12:06:46Z","abstract_excerpt":"Recently, text-to-speech (TTS) models such as FastSpeech and ParaNet have been proposed to generate mel-spectrograms from text in parallel. Despite the advantage, the parallel TTS models cannot be trained without guidance from autoregressive TTS models as their external aligners. In this work, we propose Glow-TTS, a flow-based generative model for parallel TTS that does not require any external aligner. By combining the properties of flows and dynamic programming, the proposed model searches for the most probable monotonic alignment between text and the latent representation of speech on its o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.11129","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.11129/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.11129","created_at":"2026-07-05T01:45:27.073445+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.11129v2","created_at":"2026-07-05T01:45:27.073445+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.11129","created_at":"2026-07-05T01:45:27.073445+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWTD3DMTKGQI","created_at":"2026-07-05T01:45:27.073445+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWTD3DMTKGQISEUW","created_at":"2026-07-05T01:45:27.073445+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWTD3DMT","created_at":"2026-07-05T01:45:27.073445+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19797","citing_title":"Enhancing ASR Performance in the Medical Domain for Dravidian Languages","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA","json":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA.json","graph_json":"https://pith.science/api/pith-number/TWTD3DMTKGQISEUWJ5GOJWNCPA/graph.json","events_json":"https://pith.science/api/pith-number/TWTD3DMTKGQISEUWJ5GOJWNCPA/events.json","paper":"https://pith.science/paper/TWTD3DMT"},"agent_actions":{"view_html":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA","download_json":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA.json","view_paper":"https://pith.science/paper/TWTD3DMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.11129&json=true","fetch_graph":"https://pith.science/api/pith-number/TWTD3DMTKGQISEUWJ5GOJWNCPA/graph.json","fetch_events":"https://pith.science/api/pith-number/TWTD3DMTKGQISEUWJ5GOJWNCPA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA/action/storage_attestation","attest_author":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA/action/author_attestation","sign_citation":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA/action/citation_signature","submit_replication":"https://pith.science/pith/TWTD3DMTKGQISEUWJ5GOJWNCPA/action/replication_record"}},"created_at":"2026-07-05T01:45:27.073445+00:00","updated_at":"2026-07-05T01:45:27.073445+00:00"}