{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:74QY4PCOP4FZRUP4RWG3K6QOPR","short_pith_number":"pith:74QY4PCO","schema_version":"1.0","canonical_sha256":"ff218e3c4e7f0b98d1fc8d8db57a0e7c68019eb227cb13e445c53c4b1611a5c4","source":{"kind":"arxiv","id":"2406.05965","version":1},"attestation_state":"computed","paper":{"title":"MakeSinger: A Semi-Supervised Training Method for Data-Efficient Singing Voice Synthesis via Classifier-free Diffusion Guidance","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"eess.AS","authors_text":"Byoung Jin Choi, Hyeonseung Lee, Minchan Kim, Myeonghun Jeong, Nam Soo Kim, Semin Kim","submitted_at":"2024-06-10T01:47:52Z","abstract_excerpt":"In this paper, we propose MakeSinger, a semi-supervised training method for singing voice synthesis (SVS) via classifier-free diffusion guidance. The challenge in SVS lies in the costly process of gathering aligned sets of text, pitch, and audio data. MakeSinger enables the training of the diffusion-based SVS model from any speech and singing voice data regardless of its labeling, thereby enhancing the quality of generated voices with large amount of unlabeled data. At inference, our novel dual guiding mechanism gives text and pitch guidance on the reverse diffusion step by estimating the scor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05965","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"eess.AS","submitted_at":"2024-06-10T01:47:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"04551e88707289f2511e1979bcf54bb7064791df6b037edb1e8dbeee183b11de","abstract_canon_sha256":"4a91b9d16ecc26a53ddce5fb123c594aba745b4495276f6c79770bb57749ccaf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:35.882553Z","signature_b64":"JhbTVYTTxIVLIpshoczRH/9Pq2raavG9zc4OXxKJfSP8d4KhWkKlaxrTMdJqkT6EIs7Tfjh+e7t08PHGgdPLDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff218e3c4e7f0b98d1fc8d8db57a0e7c68019eb227cb13e445c53c4b1611a5c4","last_reissued_at":"2026-07-05T08:29:35.882118Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:35.882118Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MakeSinger: A Semi-Supervised Training Method for Data-Efficient Singing Voice Synthesis via Classifier-free Diffusion Guidance","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"eess.AS","authors_text":"Byoung Jin Choi, Hyeonseung Lee, Minchan Kim, Myeonghun Jeong, Nam Soo Kim, Semin Kim","submitted_at":"2024-06-10T01:47:52Z","abstract_excerpt":"In this paper, we propose MakeSinger, a semi-supervised training method for singing voice synthesis (SVS) via classifier-free diffusion guidance. The challenge in SVS lies in the costly process of gathering aligned sets of text, pitch, and audio data. MakeSinger enables the training of the diffusion-based SVS model from any speech and singing voice data regardless of its labeling, thereby enhancing the quality of generated voices with large amount of unlabeled data. At inference, our novel dual guiding mechanism gives text and pitch guidance on the reverse diffusion step by estimating the scor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05965","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05965/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05965","created_at":"2026-07-05T08:29:35.882166+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05965v1","created_at":"2026-07-05T08:29:35.882166+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05965","created_at":"2026-07-05T08:29:35.882166+00:00"},{"alias_kind":"pith_short_12","alias_value":"74QY4PCOP4FZ","created_at":"2026-07-05T08:29:35.882166+00:00"},{"alias_kind":"pith_short_16","alias_value":"74QY4PCOP4FZRUP4","created_at":"2026-07-05T08:29:35.882166+00:00"},{"alias_kind":"pith_short_8","alias_value":"74QY4PCO","created_at":"2026-07-05T08:29:35.882166+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05852","citing_title":"UniVoice: A Unified Model for Speech and Singing Voice Generation","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR","json":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR.json","graph_json":"https://pith.science/api/pith-number/74QY4PCOP4FZRUP4RWG3K6QOPR/graph.json","events_json":"https://pith.science/api/pith-number/74QY4PCOP4FZRUP4RWG3K6QOPR/events.json","paper":"https://pith.science/paper/74QY4PCO"},"agent_actions":{"view_html":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR","download_json":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR.json","view_paper":"https://pith.science/paper/74QY4PCO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05965&json=true","fetch_graph":"https://pith.science/api/pith-number/74QY4PCOP4FZRUP4RWG3K6QOPR/graph.json","fetch_events":"https://pith.science/api/pith-number/74QY4PCOP4FZRUP4RWG3K6QOPR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR/action/storage_attestation","attest_author":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR/action/author_attestation","sign_citation":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR/action/citation_signature","submit_replication":"https://pith.science/pith/74QY4PCOP4FZRUP4RWG3K6QOPR/action/replication_record"}},"created_at":"2026-07-05T08:29:35.882166+00:00","updated_at":"2026-07-05T08:29:35.882166+00:00"}