{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:B3KVZEIQHQ6MMOEVYZ2GCDYOON","short_pith_number":"pith:B3KVZEIQ","canonical_record":{"source":{"id":"1812.04342","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-11T12:00:06Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"79a2ee2437f598e45404f86c29481939075da4eae6e6fe8ca1c1bd931fd65722","abstract_canon_sha256":"f972ef7543539af6039caebc18f943d293b22393605cccffbf4fa4364957f630"},"schema_version":"1.0"},"canonical_sha256":"0ed55c91103c3cc63895c674610f0e73758d0e8119b50010bc13d7bb59d895ce","source":{"kind":"arxiv","id":"1812.04342","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1812.04342","created_at":"2026-05-17T23:54:02Z"},{"alias_kind":"arxiv_version","alias_value":"1812.04342v2","created_at":"2026-05-17T23:54:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1812.04342","created_at":"2026-05-17T23:54:02Z"},{"alias_kind":"pith_short_12","alias_value":"B3KVZEIQHQ6M","created_at":"2026-05-18T12:32:13Z"},{"alias_kind":"pith_short_16","alias_value":"B3KVZEIQHQ6MMOEV","created_at":"2026-05-18T12:32:13Z"},{"alias_kind":"pith_short_8","alias_value":"B3KVZEIQ","created_at":"2026-05-18T12:32:13Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:B3KVZEIQHQ6MMOEVYZ2GCDYOON","target":"record","payload":{"canonical_record":{"source":{"id":"1812.04342","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-11T12:00:06Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"79a2ee2437f598e45404f86c29481939075da4eae6e6fe8ca1c1bd931fd65722","abstract_canon_sha256":"f972ef7543539af6039caebc18f943d293b22393605cccffbf4fa4364957f630"},"schema_version":"1.0"},"canonical_sha256":"0ed55c91103c3cc63895c674610f0e73758d0e8119b50010bc13d7bb59d895ce","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:54:02.276012Z","signature_b64":"lXK1G+JBGLzN1s09CvNifnq95j5DSTBGROttUUOv3xmRhbvByhtW180dJGaf0OjjvGgI404++vklNU3brSl2BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ed55c91103c3cc63895c674610f0e73758d0e8119b50010bc13d7bb59d895ce","last_reissued_at":"2026-05-17T23:54:02.275580Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:54:02.275580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1812.04342","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:54:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"H4xxMdsFnsie3lIbp6rvB0YxNTQBw8eYeNhbvK8RJrK5eusU8VwheZzrBOhruM6Lstzi25I4DQ+Mw/3x0jYuBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-28T10:05:43.999847Z"},"content_sha256":"7e882d4bc9c5c11b7968c92fd54d877cb8a87eb900ecb3a8b93a86d1101d98a3","schema_version":"1.0","event_id":"sha256:7e882d4bc9c5c11b7968c92fd54d877cb8a87eb900ecb3a8b93a86d1101d98a3"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:B3KVZEIQHQ6MMOEVYZ2GCDYOON","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Learning latent representations for style control and transfer in end-to-end speech synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Lei He, Shifeng Pan, Ya-Jie Zhang, Zhen-Hua Ling","submitted_at":"2018-12-11T12:00:06Z","abstract_excerpt":"In this paper, we introduce the Variational Autoencoder (VAE) to an end-to-end speech synthesis model, to learn the latent representation of speaking styles in an unsupervised manner. The style representation learned through VAE shows good properties such as disentangling, scaling, and combination, which makes it easy for style control. Style transfer can be achieved in this framework by first inferring style representation through the recognition network of VAE, then feeding it into TTS network to guide the style in synthesizing speech. To avoid Kullback-Leibler (KL) divergence collapse in tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1812.04342","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-17T23:54:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"K7aeW+ArDpJRrdItTfA9wq5Stno/hisKqTHiLN24zGUxNRH8zaUpAMg65SPc7EI32XhAM1jODYnEt/ToLfr3DA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-28T10:05:44.000196Z"},"content_sha256":"26e7f88093e32a29e00e0e61c9dd637e726214e0ebc912d94d5cc7abea8c03a5","schema_version":"1.0","event_id":"sha256:26e7f88093e32a29e00e0e61c9dd637e726214e0ebc912d94d5cc7abea8c03a5"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON/bundle.json","state_url":"https://pith.science/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-28T10:05:44Z","links":{"resolver":"https://pith.science/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON","bundle":"https://pith.science/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON/bundle.json","state":"https://pith.science/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON/state.json","well_known_bundle":"https://pith.science/.well-known/pith/B3KVZEIQHQ6MMOEVYZ2GCDYOON/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:B3KVZEIQHQ6MMOEVYZ2GCDYOON","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f972ef7543539af6039caebc18f943d293b22393605cccffbf4fa4364957f630","cross_cats_sorted":["cs.SD","eess.AS"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-11T12:00:06Z","title_canon_sha256":"79a2ee2437f598e45404f86c29481939075da4eae6e6fe8ca1c1bd931fd65722"},"schema_version":"1.0","source":{"id":"1812.04342","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1812.04342","created_at":"2026-05-17T23:54:02Z"},{"alias_kind":"arxiv_version","alias_value":"1812.04342v2","created_at":"2026-05-17T23:54:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1812.04342","created_at":"2026-05-17T23:54:02Z"},{"alias_kind":"pith_short_12","alias_value":"B3KVZEIQHQ6M","created_at":"2026-05-18T12:32:13Z"},{"alias_kind":"pith_short_16","alias_value":"B3KVZEIQHQ6MMOEV","created_at":"2026-05-18T12:32:13Z"},{"alias_kind":"pith_short_8","alias_value":"B3KVZEIQ","created_at":"2026-05-18T12:32:13Z"}],"graph_snapshots":[{"event_id":"sha256:26e7f88093e32a29e00e0e61c9dd637e726214e0ebc912d94d5cc7abea8c03a5","target":"graph","created_at":"2026-05-17T23:54:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"In this paper, we introduce the Variational Autoencoder (VAE) to an end-to-end speech synthesis model, to learn the latent representation of speaking styles in an unsupervised manner. The style representation learned through VAE shows good properties such as disentangling, scaling, and combination, which makes it easy for style control. Style transfer can be achieved in this framework by first inferring style representation through the recognition network of VAE, then feeding it into TTS network to guide the style in synthesizing speech. To avoid Kullback-Leibler (KL) divergence collapse in tr","authors_text":"Lei He, Shifeng Pan, Ya-Jie Zhang, Zhen-Hua Ling","cross_cats":["cs.SD","eess.AS"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-11T12:00:06Z","title":"Learning latent representations for style control and transfer in end-to-end speech synthesis"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1812.04342","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7e882d4bc9c5c11b7968c92fd54d877cb8a87eb900ecb3a8b93a86d1101d98a3","target":"record","created_at":"2026-05-17T23:54:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f972ef7543539af6039caebc18f943d293b22393605cccffbf4fa4364957f630","cross_cats_sorted":["cs.SD","eess.AS"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-12-11T12:00:06Z","title_canon_sha256":"79a2ee2437f598e45404f86c29481939075da4eae6e6fe8ca1c1bd931fd65722"},"schema_version":"1.0","source":{"id":"1812.04342","kind":"arxiv","version":2}},"canonical_sha256":"0ed55c91103c3cc63895c674610f0e73758d0e8119b50010bc13d7bb59d895ce","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0ed55c91103c3cc63895c674610f0e73758d0e8119b50010bc13d7bb59d895ce","first_computed_at":"2026-05-17T23:54:02.275580Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-17T23:54:02.275580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"lXK1G+JBGLzN1s09CvNifnq95j5DSTBGROttUUOv3xmRhbvByhtW180dJGaf0OjjvGgI404++vklNU3brSl2BA==","signature_status":"signed_v1","signed_at":"2026-05-17T23:54:02.276012Z","signed_message":"canonical_sha256_bytes"},"source_id":"1812.04342","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:7e882d4bc9c5c11b7968c92fd54d877cb8a87eb900ecb3a8b93a86d1101d98a3","sha256:26e7f88093e32a29e00e0e61c9dd637e726214e0ebc912d94d5cc7abea8c03a5"],"state_sha256":"3ab07bfd0619b3fcf7129940cf3253fb2811cb685c08e16b42ee1108db32fb57"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1gLWMixjWQvmb6EFnJQrf+zOQPzKoUafaCfsKIY+r21fdsI89yDrisoWE1hhaIFhCAzaw8UAP0fVcMKlAMMrCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-28T10:05:44.002180Z","bundle_sha256":"c88aac206cefca863151cfdd8730d857f044d89f9731a749699e1fe10ff8ac84"}}