{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OXWRS5FDHI5436XWJARXPAMADF","short_pith_number":"pith:OXWRS5FD","schema_version":"1.0","canonical_sha256":"75ed1974a33a3bcdfaf64823778180197a5c69a0c80c9de74ab50e75aee476ee","source":{"kind":"arxiv","id":"2303.13336","version":2},"attestation_state":"computed","paper":{"title":"A Survey on Audio Diffusion Models: Text To Speech Synthesis and Enhancement in Generative AI","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chaoning Zhang, Chenshuang Zhang, In So Kweon, Maryam Qamar, Mengchun Zhang, Sheng Zheng, Sung-Ho Bae","submitted_at":"2023-03-23T15:17:15Z","abstract_excerpt":"Generative AI has demonstrated impressive performance in various fields, among which speech synthesis is an interesting direction. With the diffusion model as the most popular generative model, numerous works have attempted two active tasks: text to speech and speech enhancement. This work conducts a survey on audio diffusion model, which is complementary to existing surveys that either lack the recent progress of diffusion-based speech synthesis or highlight an overall picture of applying diffusion model in multiple fields. Specifically, this work first briefly introduces the background of au"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.13336","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2023-03-23T15:17:15Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"4ddede26293c84d2d5dcdb23fca03eaf1609e541fe76ba8677253c76487362bf","abstract_canon_sha256":"0f995ca635f68359e29f0ef966256ebed8fb519ca287acef0dbc38125c78c813"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:57:16.862556Z","signature_b64":"yEbXnxz9hQiHhUd/SvKAf/77ahBf0hlDnHxP7sc4q8aqLYIT8wnUXJOezREc8L1fcNjCJZhBtPTieNbzhPrGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75ed1974a33a3bcdfaf64823778180197a5c69a0c80c9de74ab50e75aee476ee","last_reissued_at":"2026-07-05T05:57:16.862013Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:57:16.862013Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Audio Diffusion Models: Text To Speech Synthesis and Enhancement in Generative AI","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Chaoning Zhang, Chenshuang Zhang, In So Kweon, Maryam Qamar, Mengchun Zhang, Sheng Zheng, Sung-Ho Bae","submitted_at":"2023-03-23T15:17:15Z","abstract_excerpt":"Generative AI has demonstrated impressive performance in various fields, among which speech synthesis is an interesting direction. With the diffusion model as the most popular generative model, numerous works have attempted two active tasks: text to speech and speech enhancement. This work conducts a survey on audio diffusion model, which is complementary to existing surveys that either lack the recent progress of diffusion-based speech synthesis or highlight an overall picture of applying diffusion model in multiple fields. Specifically, this work first briefly introduces the background of au"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.13336","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.13336/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.13336","created_at":"2026-07-05T05:57:16.862074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.13336v2","created_at":"2026-07-05T05:57:16.862074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.13336","created_at":"2026-07-05T05:57:16.862074+00:00"},{"alias_kind":"pith_short_12","alias_value":"OXWRS5FDHI54","created_at":"2026-07-05T05:57:16.862074+00:00"},{"alias_kind":"pith_short_16","alias_value":"OXWRS5FDHI5436XW","created_at":"2026-07-05T05:57:16.862074+00:00"},{"alias_kind":"pith_short_8","alias_value":"OXWRS5FD","created_at":"2026-07-05T05:57:16.862074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25424","citing_title":"Adaptive Oscillatory Inductive Bias for Modeling Sharp Prosodic Dynamics in Diffusion-Based TTS","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00673","citing_title":"T-CLIP: Enabling Thermal Perception for Contrastive Language-Image Pretraining","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02973","citing_title":"Structured Diffusion Bridges: Inductive Bias for Denoising Diffusion Bridges","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09439","citing_title":"Inverse Design for Conditional Distribution Matching","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02973","citing_title":"Structured Diffusion Bridges: Inductive Bias for Denoising Diffusion Bridges","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10074","citing_title":"Transformers Learn the Optimal DDPM Denoiser for Multi-Token GMMs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17673","citing_title":"Grokking of Diffusion Models: Case Study on Modular Addition","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF","json":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF.json","graph_json":"https://pith.science/api/pith-number/OXWRS5FDHI5436XWJARXPAMADF/graph.json","events_json":"https://pith.science/api/pith-number/OXWRS5FDHI5436XWJARXPAMADF/events.json","paper":"https://pith.science/paper/OXWRS5FD"},"agent_actions":{"view_html":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF","download_json":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF.json","view_paper":"https://pith.science/paper/OXWRS5FD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.13336&json=true","fetch_graph":"https://pith.science/api/pith-number/OXWRS5FDHI5436XWJARXPAMADF/graph.json","fetch_events":"https://pith.science/api/pith-number/OXWRS5FDHI5436XWJARXPAMADF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF/action/storage_attestation","attest_author":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF/action/author_attestation","sign_citation":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF/action/citation_signature","submit_replication":"https://pith.science/pith/OXWRS5FDHI5436XWJARXPAMADF/action/replication_record"}},"created_at":"2026-07-05T05:57:16.862074+00:00","updated_at":"2026-07-05T05:57:16.862074+00:00"}