{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:ECJEI5H6RNQRL6DVE35LWMKRIV","short_pith_number":"pith:ECJEI5H6","schema_version":"1.0","canonical_sha256":"20924474fe8b6115f87526fabb3151456703a582ed229d28d738b257c56d8c11","source":{"kind":"arxiv","id":"2008.05289","version":1},"attestation_state":"computed","paper":{"title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Dipjyoti Paul, Yannis Pantazis, Yannis Stylianou","submitted_at":"2020-08-09T13:54:46Z","abstract_excerpt":"Recent advancements in deep learning led to human-level performance in single-speaker speech synthesis. However, there are still limitations in terms of speech quality when generalizing those systems into multiple-speaker models especially for unseen speakers and unseen recording qualities. For instance, conventional neural vocoders are adjusted to the training speaker and have poor generalization capabilities to unseen speakers. In this work, we propose a variant of WaveRNN, referred to as speaker conditional WaveRNN (SC-WaveRNN). We target towards the development of an efficient universal vo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.05289","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-08-09T13:54:46Z","cross_cats_sorted":["cs.LG","cs.SD"],"title_canon_sha256":"85b56c5d642d9f622f2923af0ec6300760b2199baf5724339b1f3290a3ea2e4b","abstract_canon_sha256":"df5d86b7185ce5f8c4964a4ed929764811de312126d31d4c408b10c8815546b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:26:45.323948Z","signature_b64":"JmR1QoWWEEF7uvpvNFqSNyvAhP282Q+mkj2oqH+0eUBmsEDM5m2kb8+A8yv97rkmr1eVBmRpKirhOGucpSY1Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20924474fe8b6115f87526fabb3151456703a582ed229d28d738b257c56d8c11","last_reissued_at":"2026-07-05T01:26:45.323541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:26:45.323541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Dipjyoti Paul, Yannis Pantazis, Yannis Stylianou","submitted_at":"2020-08-09T13:54:46Z","abstract_excerpt":"Recent advancements in deep learning led to human-level performance in single-speaker speech synthesis. However, there are still limitations in terms of speech quality when generalizing those systems into multiple-speaker models especially for unseen speakers and unseen recording qualities. For instance, conventional neural vocoders are adjusted to the training speaker and have poor generalization capabilities to unseen speakers. In this work, we propose a variant of WaveRNN, referred to as speaker conditional WaveRNN (SC-WaveRNN). We target towards the development of an efficient universal vo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.05289","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.05289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.05289","created_at":"2026-07-05T01:26:45.323613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.05289v1","created_at":"2026-07-05T01:26:45.323613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.05289","created_at":"2026-07-05T01:26:45.323613+00:00"},{"alias_kind":"pith_short_12","alias_value":"ECJEI5H6RNQR","created_at":"2026-07-05T01:26:45.323613+00:00"},{"alias_kind":"pith_short_16","alias_value":"ECJEI5H6RNQRL6DV","created_at":"2026-07-05T01:26:45.323613+00:00"},{"alias_kind":"pith_short_8","alias_value":"ECJEI5H6","created_at":"2026-07-05T01:26:45.323613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.08312","citing_title":"A Unified Model For Voice and Accent Conversion In Speech and Singing using Self-Supervised Learning and Feature Extraction","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV","json":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV.json","graph_json":"https://pith.science/api/pith-number/ECJEI5H6RNQRL6DVE35LWMKRIV/graph.json","events_json":"https://pith.science/api/pith-number/ECJEI5H6RNQRL6DVE35LWMKRIV/events.json","paper":"https://pith.science/paper/ECJEI5H6"},"agent_actions":{"view_html":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV","download_json":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV.json","view_paper":"https://pith.science/paper/ECJEI5H6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.05289&json=true","fetch_graph":"https://pith.science/api/pith-number/ECJEI5H6RNQRL6DVE35LWMKRIV/graph.json","fetch_events":"https://pith.science/api/pith-number/ECJEI5H6RNQRL6DVE35LWMKRIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV/action/storage_attestation","attest_author":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV/action/author_attestation","sign_citation":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV/action/citation_signature","submit_replication":"https://pith.science/pith/ECJEI5H6RNQRL6DVE35LWMKRIV/action/replication_record"}},"created_at":"2026-07-05T01:26:45.323613+00:00","updated_at":"2026-07-05T01:26:45.323613+00:00"}