{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3IABAYUIABTLRO3LSKEAXGTYU4","short_pith_number":"pith:3IABAYUI","schema_version":"1.0","canonical_sha256":"da001062880066b8bb6b92880b9a78a7001644031d099821ba888ed1a9bc9fdc","source":{"kind":"arxiv","id":"2204.04004","version":2},"attestation_state":"computed","paper":{"title":"Hierarchical and Multi-Scale Variational Autoencoder for Diverse and Natural Non-Autoregressive Text-to-Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Jae-Sung Bae, Jinhyeok Yang, Tae-Jun Bak, Young-Sun Joo","submitted_at":"2022-04-08T11:27:50Z","abstract_excerpt":"This paper proposes a hierarchical and multi-scale variational autoencoder-based non-autoregressive text-to-speech model (HiMuV-TTS) to generate natural speech with diverse speaking styles. Recent advances in non-autoregressive TTS (NAR-TTS) models have significantly improved the inference speed and robustness of synthesized speech. However, the diversity of speaking styles and naturalness are needed to be improved. To solve this problem, we propose the HiMuV-TTS model that first determines the global-scale prosody and then determines the local-scale prosody via conditioning on the global-scal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.04004","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2022-04-08T11:27:50Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"196ee1212dcfba4f8890a2a342512774eb85b7d7a6b7642c33aee17c64492cda","abstract_canon_sha256":"155e7844272906ad8cf9d509afdb6f011ac30ccb2c723f550d20d69643973702"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:48:12.054059Z","signature_b64":"e2bQdxQDunYonkRB/kurX1CiuhG8WcVVCXRRp0EtlJfdZDOO1PhCovcnulNPHZ0USGbjOfyZy/zHYkx+WrjAAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da001062880066b8bb6b92880b9a78a7001644031d099821ba888ed1a9bc9fdc","last_reissued_at":"2026-07-05T04:48:12.053563Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:48:12.053563Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hierarchical and Multi-Scale Variational Autoencoder for Diverse and Natural Non-Autoregressive Text-to-Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Jae-Sung Bae, Jinhyeok Yang, Tae-Jun Bak, Young-Sun Joo","submitted_at":"2022-04-08T11:27:50Z","abstract_excerpt":"This paper proposes a hierarchical and multi-scale variational autoencoder-based non-autoregressive text-to-speech model (HiMuV-TTS) to generate natural speech with diverse speaking styles. Recent advances in non-autoregressive TTS (NAR-TTS) models have significantly improved the inference speed and robustness of synthesized speech. However, the diversity of speaking styles and naturalness are needed to be improved. To solve this problem, we propose the HiMuV-TTS model that first determines the global-scale prosody and then determines the local-scale prosody via conditioning on the global-scal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.04004","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.04004/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.04004","created_at":"2026-07-05T04:48:12.053618+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.04004v2","created_at":"2026-07-05T04:48:12.053618+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.04004","created_at":"2026-07-05T04:48:12.053618+00:00"},{"alias_kind":"pith_short_12","alias_value":"3IABAYUIABTL","created_at":"2026-07-05T04:48:12.053618+00:00"},{"alias_kind":"pith_short_16","alias_value":"3IABAYUIABTLRO3L","created_at":"2026-07-05T04:48:12.053618+00:00"},{"alias_kind":"pith_short_8","alias_value":"3IABAYUI","created_at":"2026-07-05T04:48:12.053618+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4","json":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4.json","graph_json":"https://pith.science/api/pith-number/3IABAYUIABTLRO3LSKEAXGTYU4/graph.json","events_json":"https://pith.science/api/pith-number/3IABAYUIABTLRO3LSKEAXGTYU4/events.json","paper":"https://pith.science/paper/3IABAYUI"},"agent_actions":{"view_html":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4","download_json":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4.json","view_paper":"https://pith.science/paper/3IABAYUI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.04004&json=true","fetch_graph":"https://pith.science/api/pith-number/3IABAYUIABTLRO3LSKEAXGTYU4/graph.json","fetch_events":"https://pith.science/api/pith-number/3IABAYUIABTLRO3LSKEAXGTYU4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4/action/storage_attestation","attest_author":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4/action/author_attestation","sign_citation":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4/action/citation_signature","submit_replication":"https://pith.science/pith/3IABAYUIABTLRO3LSKEAXGTYU4/action/replication_record"}},"created_at":"2026-07-05T04:48:12.053618+00:00","updated_at":"2026-07-05T04:48:12.053618+00:00"}