{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VLOILJ26M2DVWHDCKRDN642F3D","short_pith_number":"pith:VLOILJ26","schema_version":"1.0","canonical_sha256":"aadc85a75e66875b1c625446df7345d8ec21dc263f508822422e475232d40a35","source":{"kind":"arxiv","id":"2406.07969","version":1},"attestation_state":"computed","paper":{"title":"LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Kentaro Tachibana, Masaya Kawamura, Ryuichi Yamamoto, Takuya Hasumi, Yuma Shirahata","submitted_at":"2024-06-12T07:49:21Z","abstract_excerpt":"We introduce LibriTTS-P, a new corpus based on LibriTTS-R that includes utterance-level descriptions (i.e., prompts) of speaking style and speaker-level prompts of speaker characteristics. We employ a hybrid approach to construct prompt annotations: (1) manual annotations that capture human perceptions of speaker characteristics and (2) synthetic annotations on speaking style. Compared to existing English prompt datasets, our corpus provides more diverse prompt annotations for all speakers of LibriTTS-R. Experimental results for prompt-based controllable TTS demonstrate that the TTS model trai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07969","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-06-12T07:49:21Z","cross_cats_sorted":["cs.CL","cs.LG","cs.SD"],"title_canon_sha256":"b3e5ed8e747208cec541a6ce57037f2eb43e4fae2c851fd2c2e942a2cd1261f1","abstract_canon_sha256":"355be3a1a1e9d2c6f70e230b8f984c38cb931dc51a3de5dcc3261553ebea13a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:41.471436Z","signature_b64":"0DoqF8Glmw09BEptogGPmdUyeN93TJ1ov0D1c2BMYZ2NHpmYDD5atllg3+Bc6LnbbR1TrojANJUwzCGS1KkABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aadc85a75e66875b1c625446df7345d8ec21dc263f508822422e475232d40a35","last_reissued_at":"2026-07-05T08:30:41.470706Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:41.470706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Kentaro Tachibana, Masaya Kawamura, Ryuichi Yamamoto, Takuya Hasumi, Yuma Shirahata","submitted_at":"2024-06-12T07:49:21Z","abstract_excerpt":"We introduce LibriTTS-P, a new corpus based on LibriTTS-R that includes utterance-level descriptions (i.e., prompts) of speaking style and speaker-level prompts of speaker characteristics. We employ a hybrid approach to construct prompt annotations: (1) manual annotations that capture human perceptions of speaker characteristics and (2) synthetic annotations on speaking style. Compared to existing English prompt datasets, our corpus provides more diverse prompt annotations for all speakers of LibriTTS-R. Experimental results for prompt-based controllable TTS demonstrate that the TTS model trai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07969","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07969/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07969","created_at":"2026-07-05T08:30:41.470947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07969v1","created_at":"2026-07-05T08:30:41.470947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07969","created_at":"2026-07-05T08:30:41.470947+00:00"},{"alias_kind":"pith_short_12","alias_value":"VLOILJ26M2DV","created_at":"2026-07-05T08:30:41.470947+00:00"},{"alias_kind":"pith_short_16","alias_value":"VLOILJ26M2DVWHDC","created_at":"2026-07-05T08:30:41.470947+00:00"},{"alias_kind":"pith_short_8","alias_value":"VLOILJ26","created_at":"2026-07-05T08:30:41.470947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19209","citing_title":"FineCombo-TTS: Collaborative and Precise Controllable Speech Synthesis Using Text Descriptions and Reference Speech","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D","json":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D.json","graph_json":"https://pith.science/api/pith-number/VLOILJ26M2DVWHDCKRDN642F3D/graph.json","events_json":"https://pith.science/api/pith-number/VLOILJ26M2DVWHDCKRDN642F3D/events.json","paper":"https://pith.science/paper/VLOILJ26"},"agent_actions":{"view_html":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D","download_json":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D.json","view_paper":"https://pith.science/paper/VLOILJ26","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07969&json=true","fetch_graph":"https://pith.science/api/pith-number/VLOILJ26M2DVWHDCKRDN642F3D/graph.json","fetch_events":"https://pith.science/api/pith-number/VLOILJ26M2DVWHDCKRDN642F3D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D/action/storage_attestation","attest_author":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D/action/author_attestation","sign_citation":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D/action/citation_signature","submit_replication":"https://pith.science/pith/VLOILJ26M2DVWHDCKRDN642F3D/action/replication_record"}},"created_at":"2026-07-05T08:30:41.470947+00:00","updated_at":"2026-07-05T08:30:41.470947+00:00"}