{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2B36X3ALDWR4PGJFLFVQQQ3CVX","short_pith_number":"pith:2B36X3AL","schema_version":"1.0","canonical_sha256":"d077ebec0b1da3c79925596b084362adc1606389835e7057e5ad14d7ef47724f","source":{"kind":"arxiv","id":"2406.04633","version":1},"attestation_state":"computed","paper":{"title":"Boosting Diffusion Model for Spectrogram Up-sampling in Text-to-speech: An Empirical Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Chong Zhang, Sheng Zhao, Yang Zheng, Yanqing Liu","submitted_at":"2024-06-07T04:34:03Z","abstract_excerpt":"Scaling text-to-speech (TTS) with autoregressive language model (LM) to large-scale datasets by quantizing waveform into discrete speech tokens is making great progress to capture the diversity and expressiveness in human speech, but the speech reconstruction quality from discrete speech token is far from satisfaction depending on the compressed speech token compression ratio. Generative diffusion models trained with score-matching loss and continuous normalized flow trained with flow-matching loss have become prominent in generation of images as well as speech. LM based TTS systems usually qu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04633","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2024-06-07T04:34:03Z","cross_cats_sorted":[],"title_canon_sha256":"0c55fdd897018d8f16894bcc0f547da00fe05b96b514976cbddec23dfc70d5d3","abstract_canon_sha256":"ca58a863311613259e2efaeae616fb77d3f2deb873ddd9d7a6cf140c3a662059"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:46.771803Z","signature_b64":"eLSDgWhI/BLoJSCvI7L/ASDkgGA7TKD9VbNhq68KHQELWB///QnuDCYazjKnxmCD5Qi8P1WEy+vrHJSVyAzfAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d077ebec0b1da3c79925596b084362adc1606389835e7057e5ad14d7ef47724f","last_reissued_at":"2026-07-05T08:28:46.771370Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:46.771370Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Boosting Diffusion Model for Spectrogram Up-sampling in Text-to-speech: An Empirical Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Chong Zhang, Sheng Zhao, Yang Zheng, Yanqing Liu","submitted_at":"2024-06-07T04:34:03Z","abstract_excerpt":"Scaling text-to-speech (TTS) with autoregressive language model (LM) to large-scale datasets by quantizing waveform into discrete speech tokens is making great progress to capture the diversity and expressiveness in human speech, but the speech reconstruction quality from discrete speech token is far from satisfaction depending on the compressed speech token compression ratio. Generative diffusion models trained with score-matching loss and continuous normalized flow trained with flow-matching loss have become prominent in generation of images as well as speech. LM based TTS systems usually qu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04633","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04633/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04633","created_at":"2026-07-05T08:28:46.771427+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04633v1","created_at":"2026-07-05T08:28:46.771427+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04633","created_at":"2026-07-05T08:28:46.771427+00:00"},{"alias_kind":"pith_short_12","alias_value":"2B36X3ALDWR4","created_at":"2026-07-05T08:28:46.771427+00:00"},{"alias_kind":"pith_short_16","alias_value":"2B36X3ALDWR4PGJF","created_at":"2026-07-05T08:28:46.771427+00:00"},{"alias_kind":"pith_short_8","alias_value":"2B36X3AL","created_at":"2026-07-05T08:28:46.771427+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.22746","citing_title":"Next Tokens Denoising for Speech Synthesis","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX","json":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX.json","graph_json":"https://pith.science/api/pith-number/2B36X3ALDWR4PGJFLFVQQQ3CVX/graph.json","events_json":"https://pith.science/api/pith-number/2B36X3ALDWR4PGJFLFVQQQ3CVX/events.json","paper":"https://pith.science/paper/2B36X3AL"},"agent_actions":{"view_html":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX","download_json":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX.json","view_paper":"https://pith.science/paper/2B36X3AL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04633&json=true","fetch_graph":"https://pith.science/api/pith-number/2B36X3ALDWR4PGJFLFVQQQ3CVX/graph.json","fetch_events":"https://pith.science/api/pith-number/2B36X3ALDWR4PGJFLFVQQQ3CVX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX/action/storage_attestation","attest_author":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX/action/author_attestation","sign_citation":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX/action/citation_signature","submit_replication":"https://pith.science/pith/2B36X3ALDWR4PGJFLFVQQQ3CVX/action/replication_record"}},"created_at":"2026-07-05T08:28:46.771427+00:00","updated_at":"2026-07-05T08:28:46.771427+00:00"}