{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KB7CSYNVHUZD5FNTSDNX25Z7MQ","short_pith_number":"pith:KB7CSYNV","schema_version":"1.0","canonical_sha256":"507e2961b53d323e95b390db7d773f641aa552619390a0a229f197c1b0092799","source":{"kind":"arxiv","id":"2412.16977","version":1},"attestation_state":"computed","paper":{"title":"Incremental Disentanglement for Environment-Aware Zero-Shot Text-to-Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Hui-Peng Du, Yang Ai, Ye-Xin Lu, Zheng-Yan Sheng, Zhen-Hua Ling","submitted_at":"2024-12-22T11:26:58Z","abstract_excerpt":"This paper proposes an Incremental Disentanglement-based Environment-Aware zero-shot text-to-speech (TTS) method, dubbed IDEA-TTS, that can synthesize speech for unseen speakers while preserving the acoustic characteristics of a given environment reference speech. IDEA-TTS adopts VITS as the TTS backbone. To effectively disentangle the environment, speaker, and text factors, we propose an incremental disentanglement process, where an environment estimator is designed to first decompose the environmental spectrogram into an environment mask and an enhanced spectrogram. The environment mask is t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16977","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-12-22T11:26:58Z","cross_cats_sorted":[],"title_canon_sha256":"911bdc6fcb21b5a3fd165d7c5e4d62d3b27accab80e0961ba49e1a1f721d1c76","abstract_canon_sha256":"ec64d72a3bce5c6c291d5530db763f1556c1e91ce85dae902e7e1f964d083834"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:15.898833Z","signature_b64":"mGPtQBXY8bDnkZW8ic3sieWOj5u8UO8qlB35Ud8SFqmljgy/YuO1mQdy4lGN9RxU+EzBu82W9Q4k5rpgyhWMDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"507e2961b53d323e95b390db7d773f641aa552619390a0a229f197c1b0092799","last_reissued_at":"2026-07-05T09:53:15.898271Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:15.898271Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Incremental Disentanglement for Environment-Aware Zero-Shot Text-to-Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Hui-Peng Du, Yang Ai, Ye-Xin Lu, Zheng-Yan Sheng, Zhen-Hua Ling","submitted_at":"2024-12-22T11:26:58Z","abstract_excerpt":"This paper proposes an Incremental Disentanglement-based Environment-Aware zero-shot text-to-speech (TTS) method, dubbed IDEA-TTS, that can synthesize speech for unseen speakers while preserving the acoustic characteristics of a given environment reference speech. IDEA-TTS adopts VITS as the TTS backbone. To effectively disentangle the environment, speaker, and text factors, we propose an incremental disentanglement process, where an environment estimator is designed to first decompose the environmental spectrogram into an environment mask and an enhanced spectrogram. The environment mask is t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16977","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16977/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16977","created_at":"2026-07-05T09:53:15.898337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16977v1","created_at":"2026-07-05T09:53:15.898337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16977","created_at":"2026-07-05T09:53:15.898337+00:00"},{"alias_kind":"pith_short_12","alias_value":"KB7CSYNVHUZD","created_at":"2026-07-05T09:53:15.898337+00:00"},{"alias_kind":"pith_short_16","alias_value":"KB7CSYNVHUZD5FNT","created_at":"2026-07-05T09:53:15.898337+00:00"},{"alias_kind":"pith_short_8","alias_value":"KB7CSYNV","created_at":"2026-07-05T09:53:15.898337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.07036","citing_title":"In This Environment, As That Speaker: A Text-Driven Framework for Multi-Attribute Speech Conversion","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ","json":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ.json","graph_json":"https://pith.science/api/pith-number/KB7CSYNVHUZD5FNTSDNX25Z7MQ/graph.json","events_json":"https://pith.science/api/pith-number/KB7CSYNVHUZD5FNTSDNX25Z7MQ/events.json","paper":"https://pith.science/paper/KB7CSYNV"},"agent_actions":{"view_html":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ","download_json":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ.json","view_paper":"https://pith.science/paper/KB7CSYNV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16977&json=true","fetch_graph":"https://pith.science/api/pith-number/KB7CSYNVHUZD5FNTSDNX25Z7MQ/graph.json","fetch_events":"https://pith.science/api/pith-number/KB7CSYNVHUZD5FNTSDNX25Z7MQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ/action/storage_attestation","attest_author":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ/action/author_attestation","sign_citation":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ/action/citation_signature","submit_replication":"https://pith.science/pith/KB7CSYNVHUZD5FNTSDNX25Z7MQ/action/replication_record"}},"created_at":"2026-07-05T09:53:15.898337+00:00","updated_at":"2026-07-05T09:53:15.898337+00:00"}