{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:DSG372VHW6SBOIKP3X6UXV34CF","short_pith_number":"pith:DSG372VH","schema_version":"1.0","canonical_sha256":"1c8dbfeaa7b7a417214fddfd4bd77c11408b5154e5f71228cfdbeb67b930f42c","source":{"kind":"arxiv","id":"2110.03887","version":4},"attestation_state":"computed","paper":{"title":"Environment Aware Text-to-Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Daxin Tan, Guangyan Zhang, Tan Lee","submitted_at":"2021-10-08T04:19:19Z","abstract_excerpt":"This study aims at designing an environment-aware text-to-speech (TTS) system that can generate speech to suit specific acoustic environments. It is also motivated by the desire to leverage massive data of speech audio from heterogeneous sources in TTS system development. The key idea is to model the acoustic environment in speech audio as a factor of data variability and incorporate it as a condition in the process of neural network based speech synthesis. Two embedding extractors are trained with two purposely constructed datasets for characterization and disentanglement of speaker and envir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.03887","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2021-10-08T04:19:19Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"8833707019a0a4f1a93933534800802684639b9ac54fef950fc36cccf68381e5","abstract_canon_sha256":"5901b5c45d694f6a9859fd3ff6034d8528ea4cc2b26e2c052f579ef79b28a3e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:46:33.060935Z","signature_b64":"bXjtkagbx1eTACxJ2XJUzg6huwTGWJx0zqi95iHLnolkrViyLJUI2IsapZH8PZESDRefVj/evaWceTc1MWuSAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c8dbfeaa7b7a417214fddfd4bd77c11408b5154e5f71228cfdbeb67b930f42c","last_reissued_at":"2026-07-05T04:46:33.060353Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:46:33.060353Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Environment Aware Text-to-Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Daxin Tan, Guangyan Zhang, Tan Lee","submitted_at":"2021-10-08T04:19:19Z","abstract_excerpt":"This study aims at designing an environment-aware text-to-speech (TTS) system that can generate speech to suit specific acoustic environments. It is also motivated by the desire to leverage massive data of speech audio from heterogeneous sources in TTS system development. The key idea is to model the acoustic environment in speech audio as a factor of data variability and incorporate it as a condition in the process of neural network based speech synthesis. Two embedding extractors are trained with two purposely constructed datasets for characterization and disentanglement of speaker and envir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.03887","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.03887/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.03887","created_at":"2026-07-05T04:46:33.060429+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.03887v4","created_at":"2026-07-05T04:46:33.060429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.03887","created_at":"2026-07-05T04:46:33.060429+00:00"},{"alias_kind":"pith_short_12","alias_value":"DSG372VHW6SB","created_at":"2026-07-05T04:46:33.060429+00:00"},{"alias_kind":"pith_short_16","alias_value":"DSG372VHW6SBOIKP","created_at":"2026-07-05T04:46:33.060429+00:00"},{"alias_kind":"pith_short_8","alias_value":"DSG372VH","created_at":"2026-07-05T04:46:33.060429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.07036","citing_title":"In This Environment, As That Speaker: A Text-Driven Framework for Multi-Attribute Speech Conversion","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF","json":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF.json","graph_json":"https://pith.science/api/pith-number/DSG372VHW6SBOIKP3X6UXV34CF/graph.json","events_json":"https://pith.science/api/pith-number/DSG372VHW6SBOIKP3X6UXV34CF/events.json","paper":"https://pith.science/paper/DSG372VH"},"agent_actions":{"view_html":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF","download_json":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF.json","view_paper":"https://pith.science/paper/DSG372VH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.03887&json=true","fetch_graph":"https://pith.science/api/pith-number/DSG372VHW6SBOIKP3X6UXV34CF/graph.json","fetch_events":"https://pith.science/api/pith-number/DSG372VHW6SBOIKP3X6UXV34CF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF/action/storage_attestation","attest_author":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF/action/author_attestation","sign_citation":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF/action/citation_signature","submit_replication":"https://pith.science/pith/DSG372VHW6SBOIKP3X6UXV34CF/action/replication_record"}},"created_at":"2026-07-05T04:46:33.060429+00:00","updated_at":"2026-07-05T04:46:33.060429+00:00"}