{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5PERJUKB2GUSPXTFRT5O2O3JAP","short_pith_number":"pith:5PERJUKB","schema_version":"1.0","canonical_sha256":"ebc914d141d1a927de658cfaed3b6903d337933bd54e4e63a595b1d60ca4b9fb","source":{"kind":"arxiv","id":"2204.09934","version":1},"attestation_state":"computed","paper":{"title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Dan Su, Dong Yu, Jun Wang, Max W. Y. Lam, Rongjie Huang, Yi Ren, Zhou Zhao","submitted_at":"2022-04-21T07:49:09Z","abstract_excerpt":"Denoising diffusion probabilistic models (DDPMs) have recently achieved leading performances in many generative tasks. However, the inherited iterative sampling process costs hindered their applications to speech synthesis. This paper proposes FastDiff, a fast conditional diffusion model for high-quality speech synthesis. FastDiff employs a stack of time-aware location-variable convolutions of diverse receptive field patterns to efficiently model long-term time dependencies with adaptive conditions. A noise schedule predictor is also adopted to reduce the sampling steps without sacrificing the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.09934","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2022-04-21T07:49:09Z","cross_cats_sorted":["cs.LG","cs.SD"],"title_canon_sha256":"57ca724ae8cb0381478f158ae72c978693d93cdcfc53970fdf165e1de4f5da50","abstract_canon_sha256":"ccbefd41da0592a1b7f8f7fbdb1fe337ed9b7884937a2b2c1903d12f50a66ad6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:16:44.749460Z","signature_b64":"NJ74zhrvH3Nd5Mb2r4jeHUyC8HocBqeYf7nY6KFutKqHLbtYAF0NlWSlb5KM1uR/qqFmgwIvneWljfJh24KIDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ebc914d141d1a927de658cfaed3b6903d337933bd54e4e63a595b1d60ca4b9fb","last_reissued_at":"2026-07-05T04:16:44.748992Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:16:44.748992Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Dan Su, Dong Yu, Jun Wang, Max W. Y. Lam, Rongjie Huang, Yi Ren, Zhou Zhao","submitted_at":"2022-04-21T07:49:09Z","abstract_excerpt":"Denoising diffusion probabilistic models (DDPMs) have recently achieved leading performances in many generative tasks. However, the inherited iterative sampling process costs hindered their applications to speech synthesis. This paper proposes FastDiff, a fast conditional diffusion model for high-quality speech synthesis. FastDiff employs a stack of time-aware location-variable convolutions of diverse receptive field patterns to efficiently model long-term time dependencies with adaptive conditions. A noise schedule predictor is also adopted to reduce the sampling steps without sacrificing the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.09934","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.09934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.09934","created_at":"2026-07-05T04:16:44.749049+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.09934v1","created_at":"2026-07-05T04:16:44.749049+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.09934","created_at":"2026-07-05T04:16:44.749049+00:00"},{"alias_kind":"pith_short_12","alias_value":"5PERJUKB2GUS","created_at":"2026-07-05T04:16:44.749049+00:00"},{"alias_kind":"pith_short_16","alias_value":"5PERJUKB2GUSPXTF","created_at":"2026-07-05T04:16:44.749049+00:00"},{"alias_kind":"pith_short_8","alias_value":"5PERJUKB","created_at":"2026-07-05T04:16:44.749049+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02913","citing_title":"A Comparison of Generative and Discriminative Methods for Speech Enhancement: Robustness, Complexity, and Hallucination","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18344","citing_title":"Improved Sample Complexity For Diffusion Model Training Without Empirical Risk Minimizer Access","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08976","citing_title":"Score-Based Generative Modeling through Anisotropic Stochastic Partial Differential Equations","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08184","citing_title":"AT-ADD: All-Type Audio Deepfake Detection Challenge Evaluation Plan","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08828","citing_title":"Post-Hoc Guidance for Consistency Models by Joint Flow Distribution Learning","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP","json":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP.json","graph_json":"https://pith.science/api/pith-number/5PERJUKB2GUSPXTFRT5O2O3JAP/graph.json","events_json":"https://pith.science/api/pith-number/5PERJUKB2GUSPXTFRT5O2O3JAP/events.json","paper":"https://pith.science/paper/5PERJUKB"},"agent_actions":{"view_html":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP","download_json":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP.json","view_paper":"https://pith.science/paper/5PERJUKB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.09934&json=true","fetch_graph":"https://pith.science/api/pith-number/5PERJUKB2GUSPXTFRT5O2O3JAP/graph.json","fetch_events":"https://pith.science/api/pith-number/5PERJUKB2GUSPXTFRT5O2O3JAP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP/action/storage_attestation","attest_author":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP/action/author_attestation","sign_citation":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP/action/citation_signature","submit_replication":"https://pith.science/pith/5PERJUKB2GUSPXTFRT5O2O3JAP/action/replication_record"}},"created_at":"2026-07-05T04:16:44.749049+00:00","updated_at":"2026-07-05T04:16:44.749049+00:00"}