{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5TIYS4OOJUD3S2O77OOAD6B5ZO","short_pith_number":"pith:5TIYS4OO","schema_version":"1.0","canonical_sha256":"ecd18971ce4d07b969dffb9c01f83dcb9c3695ba0ddfdb34e7308a835bafe4be","source":{"kind":"arxiv","id":"2312.12181","version":1},"attestation_state":"computed","paper":{"title":"StyleSpeech: Self-supervised Style Enhancing with VQ-VAE-based Pre-training for Expressive Audiobook Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Helen Meng, Lei He, Shaofei Zhang, Xi Wang, Xixin Wu, Xueyuan Chen, Zhiyong Wu","submitted_at":"2023-12-19T14:13:26Z","abstract_excerpt":"The expressive quality of synthesized speech for audiobooks is limited by generalized model architecture and unbalanced style distribution in the training data. To address these issues, in this paper, we propose a self-supervised style enhancing method with VQ-VAE-based pre-training for expressive audiobook speech synthesis. Firstly, a text style encoder is pre-trained with a large amount of unlabeled text-only data. Secondly, a spectrogram style extractor based on VQ-VAE is pre-trained in a self-supervised manner, with plenty of audio data that covers complex style variations. Then a novel ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.12181","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-12-19T14:13:26Z","cross_cats_sorted":["cs.AI","eess.AS"],"title_canon_sha256":"32505cee97da1fc6c09f4094d339a79abc1bd4d39160e97c88e172e51bbb7bd9","abstract_canon_sha256":"e8fcb42394179bad4cad4f951d27fc453530558012d016e22d5695942d0a5d16"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:59.833352Z","signature_b64":"DQIKmtKlSiBtiQwNbRIkf/4/hGCTCd2RJCtqKLSu/BiJupM056n+mGofiZGFkuN/FSb0jYCJhtKrFQLw8r97BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecd18971ce4d07b969dffb9c01f83dcb9c3695ba0ddfdb34e7308a835bafe4be","last_reissued_at":"2026-07-05T07:25:59.832934Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:59.832934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StyleSpeech: Self-supervised Style Enhancing with VQ-VAE-based Pre-training for Expressive Audiobook Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","eess.AS"],"primary_cat":"cs.SD","authors_text":"Helen Meng, Lei He, Shaofei Zhang, Xi Wang, Xixin Wu, Xueyuan Chen, Zhiyong Wu","submitted_at":"2023-12-19T14:13:26Z","abstract_excerpt":"The expressive quality of synthesized speech for audiobooks is limited by generalized model architecture and unbalanced style distribution in the training data. To address these issues, in this paper, we propose a self-supervised style enhancing method with VQ-VAE-based pre-training for expressive audiobook speech synthesis. Firstly, a text style encoder is pre-trained with a large amount of unlabeled text-only data. Secondly, a spectrogram style extractor based on VQ-VAE is pre-trained in a self-supervised manner, with plenty of audio data that covers complex style variations. Then a novel ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.12181","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.12181/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.12181","created_at":"2026-07-05T07:25:59.832993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.12181v1","created_at":"2026-07-05T07:25:59.832993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.12181","created_at":"2026-07-05T07:25:59.832993+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TIYS4OOJUD3","created_at":"2026-07-05T07:25:59.832993+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TIYS4OOJUD3S2O7","created_at":"2026-07-05T07:25:59.832993+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TIYS4OO","created_at":"2026-07-05T07:25:59.832993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO","json":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO.json","graph_json":"https://pith.science/api/pith-number/5TIYS4OOJUD3S2O77OOAD6B5ZO/graph.json","events_json":"https://pith.science/api/pith-number/5TIYS4OOJUD3S2O77OOAD6B5ZO/events.json","paper":"https://pith.science/paper/5TIYS4OO"},"agent_actions":{"view_html":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO","download_json":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO.json","view_paper":"https://pith.science/paper/5TIYS4OO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.12181&json=true","fetch_graph":"https://pith.science/api/pith-number/5TIYS4OOJUD3S2O77OOAD6B5ZO/graph.json","fetch_events":"https://pith.science/api/pith-number/5TIYS4OOJUD3S2O77OOAD6B5ZO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO/action/storage_attestation","attest_author":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO/action/author_attestation","sign_citation":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO/action/citation_signature","submit_replication":"https://pith.science/pith/5TIYS4OOJUD3S2O77OOAD6B5ZO/action/replication_record"}},"created_at":"2026-07-05T07:25:59.832993+00:00","updated_at":"2026-07-05T07:25:59.832993+00:00"}