{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HULB5C4XTMTSACJG6TUIW7UWIY","short_pith_number":"pith:HULB5C4X","schema_version":"1.0","canonical_sha256":"3d161e8b979b27200926f4e88b7e9646139a97d1cb96f9bed26e415ea44beddf","source":{"kind":"arxiv","id":"2104.11673","version":1},"attestation_state":"computed","paper":{"title":"Deep Learning Based Assessment of Synthetic Speech Naturalness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Gabriel Mittag, Sebastian M\\\"oller","submitted_at":"2021-04-23T16:05:20Z","abstract_excerpt":"In this paper, we present a new objective prediction model for synthetic speech naturalness. It can be used to evaluate Text-To-Speech or Voice Conversion systems and works language independently. The model is trained end-to-end and based on a CNN-LSTM network that previously showed to give good results for speech quality estimation. We trained and tested the model on 16 different datasets, such as from the Blizzard Challenge and the Voice Conversion Challenge. Further, we show that the reliability of deep learning-based naturalness prediction can be improved by transfer learning from speech q"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.11673","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2021-04-23T16:05:20Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","eess.AS"],"title_canon_sha256":"6e06e78fa322406148e8cee3dd477703bc19ac991e6c3e4263ce5944baa6764d","abstract_canon_sha256":"4c2cd2dc3ee48fe592814f2fb3fb8642fade9adcf3d9e3d4e9753c05a3372e99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:34:34.583442Z","signature_b64":"g8O/jZn8/dyrM0HDkRdfW9TvaDjx+tpJQuOK3QB37BzuFe1erqjTSF2z3q7gOD/myuMuqLLVRXVu7ziIFUPWAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d161e8b979b27200926f4e88b7e9646139a97d1cb96f9bed26e415ea44beddf","last_reissued_at":"2026-07-05T02:34:34.583039Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:34:34.583039Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Learning Based Assessment of Synthetic Speech Naturalness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Gabriel Mittag, Sebastian M\\\"oller","submitted_at":"2021-04-23T16:05:20Z","abstract_excerpt":"In this paper, we present a new objective prediction model for synthetic speech naturalness. It can be used to evaluate Text-To-Speech or Voice Conversion systems and works language independently. The model is trained end-to-end and based on a CNN-LSTM network that previously showed to give good results for speech quality estimation. We trained and tested the model on 16 different datasets, such as from the Blizzard Challenge and the Voice Conversion Challenge. Further, we show that the reliability of deep learning-based naturalness prediction can be improved by transfer learning from speech q"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.11673","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.11673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.11673","created_at":"2026-07-05T02:34:34.583103+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.11673v1","created_at":"2026-07-05T02:34:34.583103+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.11673","created_at":"2026-07-05T02:34:34.583103+00:00"},{"alias_kind":"pith_short_12","alias_value":"HULB5C4XTMTS","created_at":"2026-07-05T02:34:34.583103+00:00"},{"alias_kind":"pith_short_16","alias_value":"HULB5C4XTMTSACJG","created_at":"2026-07-05T02:34:34.583103+00:00"},{"alias_kind":"pith_short_8","alias_value":"HULB5C4X","created_at":"2026-07-05T02:34:34.583103+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.00324","citing_title":"Collecting, Curating, and Annotating Good Quality Speech deepfake dataset for Famous Figures: Process and Challenges","ref_index":51,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY","json":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY.json","graph_json":"https://pith.science/api/pith-number/HULB5C4XTMTSACJG6TUIW7UWIY/graph.json","events_json":"https://pith.science/api/pith-number/HULB5C4XTMTSACJG6TUIW7UWIY/events.json","paper":"https://pith.science/paper/HULB5C4X"},"agent_actions":{"view_html":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY","download_json":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY.json","view_paper":"https://pith.science/paper/HULB5C4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.11673&json=true","fetch_graph":"https://pith.science/api/pith-number/HULB5C4XTMTSACJG6TUIW7UWIY/graph.json","fetch_events":"https://pith.science/api/pith-number/HULB5C4XTMTSACJG6TUIW7UWIY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY/action/storage_attestation","attest_author":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY/action/author_attestation","sign_citation":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY/action/citation_signature","submit_replication":"https://pith.science/pith/HULB5C4XTMTSACJG6TUIW7UWIY/action/replication_record"}},"created_at":"2026-07-05T02:34:34.583103+00:00","updated_at":"2026-07-05T02:34:34.583103+00:00"}