{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IT3N7G2WWZPU5NPEDBQUG47ZJH","short_pith_number":"pith:IT3N7G2W","schema_version":"1.0","canonical_sha256":"44f6df9b56b65f4eb5e418614373f949c98aeb0f86ad7751e98b7600af97c4b2","source":{"kind":"arxiv","id":"2305.18802","version":1},"attestation_state":"computed","paper":{"title":"LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Ankur Bapna, Heiga Zen, Kohei Yatabe, Michiel Bacchiani, Nobuyuki Morioka, Shigeki Karita, Wei Han, Yifan Ding, Yuma Koizumi, Yu Zhang","submitted_at":"2023-05-30T07:30:21Z","abstract_excerpt":"This paper introduces a new speech dataset called ``LibriTTS-R'' designed for text-to-speech (TTS) use. It is derived by applying speech restoration to the LibriTTS corpus, which consists of 585 hours of speech data at 24 kHz sampling rate from 2,456 speakers and the corresponding texts. The constituent samples of LibriTTS-R are identical to those of LibriTTS, with only the sound quality improved. Experimental results show that the LibriTTS-R ground-truth samples showed significantly improved sound quality compared to those in LibriTTS. In addition, neural end-to-end TTS trained with LibriTTS-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18802","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"eess.AS","submitted_at":"2023-05-30T07:30:21Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"0ec69e49c0b8eda80ac5420b321eae52abdf6c120f7eb50ea7219050947375d3","abstract_canon_sha256":"7e6847900d45d7a2628caa5e240a1ad01bb532f3982adc247fca200952ec63d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:05.058161Z","signature_b64":"mSQIG18lG/SrXR5ZX/e6dmoC7SfNXbjunAcaZuN7lqhnr3RJ+PXnR2lxavO4AvmmXwt5IMRJIZRMl0izMMxjCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"44f6df9b56b65f4eb5e418614373f949c98aeb0f86ad7751e98b7600af97c4b2","last_reissued_at":"2026-07-05T06:15:05.057739Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:05.057739Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Ankur Bapna, Heiga Zen, Kohei Yatabe, Michiel Bacchiani, Nobuyuki Morioka, Shigeki Karita, Wei Han, Yifan Ding, Yuma Koizumi, Yu Zhang","submitted_at":"2023-05-30T07:30:21Z","abstract_excerpt":"This paper introduces a new speech dataset called ``LibriTTS-R'' designed for text-to-speech (TTS) use. It is derived by applying speech restoration to the LibriTTS corpus, which consists of 585 hours of speech data at 24 kHz sampling rate from 2,456 speakers and the corresponding texts. The constituent samples of LibriTTS-R are identical to those of LibriTTS, with only the sound quality improved. Experimental results show that the LibriTTS-R ground-truth samples showed significantly improved sound quality compared to those in LibriTTS. In addition, neural end-to-end TTS trained with LibriTTS-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18802","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18802/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18802","created_at":"2026-07-05T06:15:05.057797+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18802v1","created_at":"2026-07-05T06:15:05.057797+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18802","created_at":"2026-07-05T06:15:05.057797+00:00"},{"alias_kind":"pith_short_12","alias_value":"IT3N7G2WWZPU","created_at":"2026-07-05T06:15:05.057797+00:00"},{"alias_kind":"pith_short_16","alias_value":"IT3N7G2WWZPU5NPE","created_at":"2026-07-05T06:15:05.057797+00:00"},{"alias_kind":"pith_short_8","alias_value":"IT3N7G2W","created_at":"2026-07-05T06:15:05.057797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22811","citing_title":"Bagpiper-TTS: Natural Language Guided Universal Speech Synthesis","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19209","citing_title":"FineCombo-TTS: Collaborative and Precise Controllable Speech Synthesis Using Text Descriptions and Reference Speech","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12940","citing_title":"Self-Guidance: Enhancing Neural Codecs via Decoder Manifold Alignment","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09553","citing_title":"OpenBibleTTS: Large-Scale Speech Resources and TTS Models for Low-Resource Languages","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05889","citing_title":"GLASS: GRPO-Trained LoRA for Acoustic Style Steering in Zero-Shot Text-to-Speech","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27741","citing_title":"Escape the Language Prior: Mitigating Late-Stage Modality Collapse in Audio Reasoning via Modality-Aware Policy Optimization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20830","citing_title":"Raon-OpenTTS: Open Models and Data for Robust Text-to-Speech","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02954","citing_title":"The World is Not Mono: Enabling Spatial Understanding in Large Audio-Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08184","citing_title":"AT-ADD: All-Type Audio Deepfake Detection Challenge Evaluation Plan","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH","json":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH.json","graph_json":"https://pith.science/api/pith-number/IT3N7G2WWZPU5NPEDBQUG47ZJH/graph.json","events_json":"https://pith.science/api/pith-number/IT3N7G2WWZPU5NPEDBQUG47ZJH/events.json","paper":"https://pith.science/paper/IT3N7G2W"},"agent_actions":{"view_html":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH","download_json":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH.json","view_paper":"https://pith.science/paper/IT3N7G2W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18802&json=true","fetch_graph":"https://pith.science/api/pith-number/IT3N7G2WWZPU5NPEDBQUG47ZJH/graph.json","fetch_events":"https://pith.science/api/pith-number/IT3N7G2WWZPU5NPEDBQUG47ZJH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH/action/storage_attestation","attest_author":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH/action/author_attestation","sign_citation":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH/action/citation_signature","submit_replication":"https://pith.science/pith/IT3N7G2WWZPU5NPEDBQUG47ZJH/action/replication_record"}},"created_at":"2026-07-05T06:15:05.057797+00:00","updated_at":"2026-07-05T06:15:05.057797+00:00"}