{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:BYS2AEVS5OFW7T4JJW2XIJNDEI","short_pith_number":"pith:BYS2AEVS","schema_version":"1.0","canonical_sha256":"0e25a012b2eb8b6fcf894db57425a32219134026e2070f3b531ff34dcc4885de","source":{"kind":"arxiv","id":"2006.03575","version":3},"attestation_state":"computed","paper":{"title":"End-to-End Adversarial Text-to-Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Erich Elsen, Jeff Donahue, Karen Simonyan, Miko{\\l}aj Bi\\'nkowski, Sander Dieleman","submitted_at":"2020-06-05T17:41:05Z","abstract_excerpt":"Modern text-to-speech synthesis pipelines typically involve multiple processing stages, each of which is designed or learnt independently from the rest. In this work, we take on the challenging task of learning to synthesise speech from normalised text or phonemes in an end-to-end manner, resulting in models which operate directly on character or phoneme input sequences and produce raw speech audio outputs. Our proposed generator is feed-forward and thus efficient for both training and inference, using a differentiable alignment scheme based on token length prediction. It learns to produce hig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.03575","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2020-06-05T17:41:05Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"fe55bf676349841e553d96e4a5d9923f24fea2a05675b8d615598afb6661c287","abstract_canon_sha256":"d00efcd576c79cc65225c67763da7bb484f84b7901244566edeedac9e7d46a27"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:23:57.124200Z","signature_b64":"x2B8Vi9rPV7N2RqdOP/+BmyU1FtBIPTSyH/YAQrpOa0se0ZGhVNHyhKKDOAL1gDVmZLGJyrktNy4UL+xxYnIDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e25a012b2eb8b6fcf894db57425a32219134026e2070f3b531ff34dcc4885de","last_reissued_at":"2026-07-05T02:23:57.123748Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:23:57.123748Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-End Adversarial Text-to-Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Erich Elsen, Jeff Donahue, Karen Simonyan, Miko{\\l}aj Bi\\'nkowski, Sander Dieleman","submitted_at":"2020-06-05T17:41:05Z","abstract_excerpt":"Modern text-to-speech synthesis pipelines typically involve multiple processing stages, each of which is designed or learnt independently from the rest. In this work, we take on the challenging task of learning to synthesise speech from normalised text or phonemes in an end-to-end manner, resulting in models which operate directly on character or phoneme input sequences and produce raw speech audio outputs. Our proposed generator is feed-forward and thus efficient for both training and inference, using a differentiable alignment scheme based on token length prediction. It learns to produce hig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.03575","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.03575/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.03575","created_at":"2026-07-05T02:23:57.123801+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.03575v3","created_at":"2026-07-05T02:23:57.123801+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.03575","created_at":"2026-07-05T02:23:57.123801+00:00"},{"alias_kind":"pith_short_12","alias_value":"BYS2AEVS5OFW","created_at":"2026-07-05T02:23:57.123801+00:00"},{"alias_kind":"pith_short_16","alias_value":"BYS2AEVS5OFW7T4J","created_at":"2026-07-05T02:23:57.123801+00:00"},{"alias_kind":"pith_short_8","alias_value":"BYS2AEVS","created_at":"2026-07-05T02:23:57.123801+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03455","citing_title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2009.09761","citing_title":"DiffWave: A Versatile Diffusion Model for Audio Synthesis","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI","json":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI.json","graph_json":"https://pith.science/api/pith-number/BYS2AEVS5OFW7T4JJW2XIJNDEI/graph.json","events_json":"https://pith.science/api/pith-number/BYS2AEVS5OFW7T4JJW2XIJNDEI/events.json","paper":"https://pith.science/paper/BYS2AEVS"},"agent_actions":{"view_html":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI","download_json":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI.json","view_paper":"https://pith.science/paper/BYS2AEVS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.03575&json=true","fetch_graph":"https://pith.science/api/pith-number/BYS2AEVS5OFW7T4JJW2XIJNDEI/graph.json","fetch_events":"https://pith.science/api/pith-number/BYS2AEVS5OFW7T4JJW2XIJNDEI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI/action/storage_attestation","attest_author":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI/action/author_attestation","sign_citation":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI/action/citation_signature","submit_replication":"https://pith.science/pith/BYS2AEVS5OFW7T4JJW2XIJNDEI/action/replication_record"}},"created_at":"2026-07-05T02:23:57.123801+00:00","updated_at":"2026-07-05T02:23:57.123801+00:00"}