{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VNERNF6OMCGQHZTEDMCGM54MHO","short_pith_number":"pith:VNERNF6O","schema_version":"1.0","canonical_sha256":"ab491697ce608d03e6641b0466778c3b8ad1abc0f8bab54e5e2bae4f6835efff","source":{"kind":"arxiv","id":"2309.13664","version":1},"attestation_state":"computed","paper":{"title":"VoiceLDM: Text-to-Speech with Environmental Context","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Inmo Yeon, Joon Son Chung, Juhan Nam, Yeonghyeon Lee","submitted_at":"2023-09-24T15:20:59Z","abstract_excerpt":"This paper presents VoiceLDM, a model designed to produce audio that accurately follows two distinct natural language text prompts: the description prompt and the content prompt. The former provides information about the overall environmental context of the audio, while the latter conveys the linguistic content. To achieve this, we adopt a text-to-audio (TTA) model based on latent diffusion models and extend its functionality to incorporate an additional content prompt as a conditional input. By utilizing pretrained contrastive language-audio pretraining (CLAP) and Whisper, VoiceLDM is trained"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.13664","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-09-24T15:20:59Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","cs.SD"],"title_canon_sha256":"8b7815f830379f78963c8d562e127dc1fdcb24047d36cf3892c6070443816fa8","abstract_canon_sha256":"81c41275bcc9a451ef0484c16405bb150385f1e5d318d8019f2014bbcb35a403"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:53:55.645016Z","signature_b64":"lSXcb3PBhS0IShxXyW3SlXL4+xukxlvkEtlIQjlyeAPyf6ap2tl3GLVRAHIBg1q/sSshYVeMr/DWmKLALdNFBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab491697ce608d03e6641b0466778c3b8ad1abc0f8bab54e5e2bae4f6835efff","last_reissued_at":"2026-07-05T06:53:55.644586Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:53:55.644586Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VoiceLDM: Text-to-Speech with Environmental Context","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Inmo Yeon, Joon Son Chung, Juhan Nam, Yeonghyeon Lee","submitted_at":"2023-09-24T15:20:59Z","abstract_excerpt":"This paper presents VoiceLDM, a model designed to produce audio that accurately follows two distinct natural language text prompts: the description prompt and the content prompt. The former provides information about the overall environmental context of the audio, while the latter conveys the linguistic content. To achieve this, we adopt a text-to-audio (TTA) model based on latent diffusion models and extend its functionality to incorporate an additional content prompt as a conditional input. By utilizing pretrained contrastive language-audio pretraining (CLAP) and Whisper, VoiceLDM is trained"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.13664","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.13664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.13664","created_at":"2026-07-05T06:53:55.644662+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.13664v1","created_at":"2026-07-05T06:53:55.644662+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.13664","created_at":"2026-07-05T06:53:55.644662+00:00"},{"alias_kind":"pith_short_12","alias_value":"VNERNF6OMCGQ","created_at":"2026-07-05T06:53:55.644662+00:00"},{"alias_kind":"pith_short_16","alias_value":"VNERNF6OMCGQHZTE","created_at":"2026-07-05T06:53:55.644662+00:00"},{"alias_kind":"pith_short_8","alias_value":"VNERNF6O","created_at":"2026-07-05T06:53:55.644662+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00927","citing_title":"In-the-wild Audio Spatialization with Flexible Text-guided Localization","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO","json":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO.json","graph_json":"https://pith.science/api/pith-number/VNERNF6OMCGQHZTEDMCGM54MHO/graph.json","events_json":"https://pith.science/api/pith-number/VNERNF6OMCGQHZTEDMCGM54MHO/events.json","paper":"https://pith.science/paper/VNERNF6O"},"agent_actions":{"view_html":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO","download_json":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO.json","view_paper":"https://pith.science/paper/VNERNF6O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.13664&json=true","fetch_graph":"https://pith.science/api/pith-number/VNERNF6OMCGQHZTEDMCGM54MHO/graph.json","fetch_events":"https://pith.science/api/pith-number/VNERNF6OMCGQHZTEDMCGM54MHO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO/action/storage_attestation","attest_author":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO/action/author_attestation","sign_citation":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO/action/citation_signature","submit_replication":"https://pith.science/pith/VNERNF6OMCGQHZTEDMCGM54MHO/action/replication_record"}},"created_at":"2026-07-05T06:53:55.644662+00:00","updated_at":"2026-07-05T06:53:55.644662+00:00"}