{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DDQR5ERQKB64OZQHPWTRLALM22","short_pith_number":"pith:DDQR5ERQ","schema_version":"1.0","canonical_sha256":"18e11e9230507dc766077da715816cd68cb6797538648ac908c907256a98b904","source":{"kind":"arxiv","id":"2507.09310","version":1},"attestation_state":"computed","paper":{"title":"Voice Conversion for Lombard Speaking Style with Implicit and Explicit Acoustic Feature Conditioning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Daniel Korzekwa, Dominika Woszczyk, Manuel Sam Ribeiro, Thomas Merritt","submitted_at":"2025-07-12T14:57:04Z","abstract_excerpt":"Text-to-Speech (TTS) systems in Lombard speaking style can improve the overall intelligibility of speech, useful for hearing loss and noisy conditions. However, training those models requires a large amount of data and the Lombard effect is challenging to record due to speaker and noise variability and tiring recording conditions. Voice conversion (VC) has been shown to be a useful augmentation technique to train TTS systems in the absence of recorded data from the target speaker in the target speaking style. In this paper, we are concerned with Lombard speaking style transfer. Our goal is to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.09310","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2025-07-12T14:57:04Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"39282c01c0af8444b7f0580707f6966a641c3c990cc645eb387146545d1c5ddf","abstract_canon_sha256":"2e67a37922dd26f2f8f16a9ce926eb2c9584fb08c726ddf09cd85e760d1b72b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:07.056998Z","signature_b64":"GnmoiSTfd4h7OE5DAOjTgu7yCcmwQ+yrZRI0QybpiYNIcnfT224HnHMzO3CyVtC4M3gNN7NSla2AgYWOuT9RAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18e11e9230507dc766077da715816cd68cb6797538648ac908c907256a98b904","last_reissued_at":"2026-07-05T11:36:07.056543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:07.056543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Voice Conversion for Lombard Speaking Style with Implicit and Explicit Acoustic Feature Conditioning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Daniel Korzekwa, Dominika Woszczyk, Manuel Sam Ribeiro, Thomas Merritt","submitted_at":"2025-07-12T14:57:04Z","abstract_excerpt":"Text-to-Speech (TTS) systems in Lombard speaking style can improve the overall intelligibility of speech, useful for hearing loss and noisy conditions. However, training those models requires a large amount of data and the Lombard effect is challenging to record due to speaker and noise variability and tiring recording conditions. Voice conversion (VC) has been shown to be a useful augmentation technique to train TTS systems in the absence of recorded data from the target speaker in the target speaking style. In this paper, we are concerned with Lombard speaking style transfer. Our goal is to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.09310","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.09310/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.09310","created_at":"2026-07-05T11:36:07.056600+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.09310v1","created_at":"2026-07-05T11:36:07.056600+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.09310","created_at":"2026-07-05T11:36:07.056600+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDQR5ERQKB64","created_at":"2026-07-05T11:36:07.056600+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDQR5ERQKB64OZQH","created_at":"2026-07-05T11:36:07.056600+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDQR5ERQ","created_at":"2026-07-05T11:36:07.056600+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23176","citing_title":"Synthesizing the Lombard Effect: Multi-Level Control of Speech Clarity and Vocal Effort in TTS","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22","json":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22.json","graph_json":"https://pith.science/api/pith-number/DDQR5ERQKB64OZQHPWTRLALM22/graph.json","events_json":"https://pith.science/api/pith-number/DDQR5ERQKB64OZQHPWTRLALM22/events.json","paper":"https://pith.science/paper/DDQR5ERQ"},"agent_actions":{"view_html":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22","download_json":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22.json","view_paper":"https://pith.science/paper/DDQR5ERQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.09310&json=true","fetch_graph":"https://pith.science/api/pith-number/DDQR5ERQKB64OZQHPWTRLALM22/graph.json","fetch_events":"https://pith.science/api/pith-number/DDQR5ERQKB64OZQHPWTRLALM22/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22/action/storage_attestation","attest_author":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22/action/author_attestation","sign_citation":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22/action/citation_signature","submit_replication":"https://pith.science/pith/DDQR5ERQKB64OZQHPWTRLALM22/action/replication_record"}},"created_at":"2026-07-05T11:36:07.056600+00:00","updated_at":"2026-07-05T11:36:07.056600+00:00"}