{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:LN4GBL7T7TS4NKYLCYT7MVUSH2","short_pith_number":"pith:LN4GBL7T","schema_version":"1.0","canonical_sha256":"5b7860aff3fce5c6ab0b1627f656923e83d4209fc22dd9e9faa9d836cac36361","source":{"kind":"arxiv","id":"2212.14518","version":1},"attestation_state":"computed","paper":{"title":"ResGrad: Residual Denoising Diffusion Probabilistic Models for Text to Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD","eess.SP"],"primary_cat":"eess.AS","authors_text":"Danilo Mandic, Haohe Liu, Jiang Bian, Jiawei Chen, Ke Wang, Lei He, Sheng Zhao, Xu Tan, Yang Cui, Yichong Leng, Yihan Wu, Zehua Chen","submitted_at":"2022-12-30T02:31:35Z","abstract_excerpt":"Denoising Diffusion Probabilistic Models (DDPMs) are emerging in text-to-speech (TTS) synthesis because of their strong capability of generating high-fidelity samples. However, their iterative refinement process in high-dimensional data space results in slow inference speed, which restricts their application in real-time systems. Previous works have explored speeding up by minimizing the number of inference steps but at the cost of sample quality. In this work, to improve the inference speed for DDPM-based TTS model while achieving high sample quality, we propose ResGrad, a lightweight diffusi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.14518","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2022-12-30T02:31:35Z","cross_cats_sorted":["cs.CL","cs.LG","cs.SD","eess.SP"],"title_canon_sha256":"2ce2020a80fc96d96fee3797a84b57abce0b08b90323a9ea917d226cecf64fd7","abstract_canon_sha256":"8a50fc1e86876587979c9450cd34bcb73ae1dcbbccc49431144193ce2dd43614"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:29:11.136663Z","signature_b64":"rIwWleevh13u0MZZrW1sejWL5JlPjTDYWdUzmjdRZjkLS0BbPogtoEPhBbgrdJTesQqvI55WIileSCdnqxKRDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5b7860aff3fce5c6ab0b1627f656923e83d4209fc22dd9e9faa9d836cac36361","last_reissued_at":"2026-07-05T05:29:11.136185Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:29:11.136185Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ResGrad: Residual Denoising Diffusion Probabilistic Models for Text to Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD","eess.SP"],"primary_cat":"eess.AS","authors_text":"Danilo Mandic, Haohe Liu, Jiang Bian, Jiawei Chen, Ke Wang, Lei He, Sheng Zhao, Xu Tan, Yang Cui, Yichong Leng, Yihan Wu, Zehua Chen","submitted_at":"2022-12-30T02:31:35Z","abstract_excerpt":"Denoising Diffusion Probabilistic Models (DDPMs) are emerging in text-to-speech (TTS) synthesis because of their strong capability of generating high-fidelity samples. However, their iterative refinement process in high-dimensional data space results in slow inference speed, which restricts their application in real-time systems. Previous works have explored speeding up by minimizing the number of inference steps but at the cost of sample quality. In this work, to improve the inference speed for DDPM-based TTS model while achieving high sample quality, we propose ResGrad, a lightweight diffusi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.14518","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.14518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.14518","created_at":"2026-07-05T05:29:11.136244+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.14518v1","created_at":"2026-07-05T05:29:11.136244+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.14518","created_at":"2026-07-05T05:29:11.136244+00:00"},{"alias_kind":"pith_short_12","alias_value":"LN4GBL7T7TS4","created_at":"2026-07-05T05:29:11.136244+00:00"},{"alias_kind":"pith_short_16","alias_value":"LN4GBL7T7TS4NKYL","created_at":"2026-07-05T05:29:11.136244+00:00"},{"alias_kind":"pith_short_8","alias_value":"LN4GBL7T","created_at":"2026-07-05T05:29:11.136244+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03119","citing_title":"GuidedBridge: Training-freely Improving Bridge Models with Prior Guidance","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30821","citing_title":"Mind the Residual Gap: Probabilistic Downscaling under Real-World Bias","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06027","citing_title":"DreamAudio: Customized Text-to-Audio Generation with Diffusion Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02430","citing_title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2","json":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2.json","graph_json":"https://pith.science/api/pith-number/LN4GBL7T7TS4NKYLCYT7MVUSH2/graph.json","events_json":"https://pith.science/api/pith-number/LN4GBL7T7TS4NKYLCYT7MVUSH2/events.json","paper":"https://pith.science/paper/LN4GBL7T"},"agent_actions":{"view_html":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2","download_json":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2.json","view_paper":"https://pith.science/paper/LN4GBL7T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.14518&json=true","fetch_graph":"https://pith.science/api/pith-number/LN4GBL7T7TS4NKYLCYT7MVUSH2/graph.json","fetch_events":"https://pith.science/api/pith-number/LN4GBL7T7TS4NKYLCYT7MVUSH2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2/action/storage_attestation","attest_author":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2/action/author_attestation","sign_citation":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2/action/citation_signature","submit_replication":"https://pith.science/pith/LN4GBL7T7TS4NKYLCYT7MVUSH2/action/replication_record"}},"created_at":"2026-07-05T05:29:11.136244+00:00","updated_at":"2026-07-05T05:29:11.136244+00:00"}