{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S5O3POKXYJAENN2HDPS2EPHRML","short_pith_number":"pith:S5O3POKX","schema_version":"1.0","canonical_sha256":"975db7b957c24046b7471be5a23cf162ef34116bb92ef4fad1d991fc94f6a383","source":{"kind":"arxiv","id":"2311.08355","version":3},"attestation_state":"computed","paper":{"title":"Mustango: Toward Controllable Text-to-Music Generation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Deepanway Ghosal, Dorien Herremans, Jan Melechovsky, Navonil Majumder, Soujanya Poria, Zixun Guo","submitted_at":"2023-11-14T17:54:38Z","abstract_excerpt":"The quality of the text-to-music models has reached new heights due to recent advancements in diffusion models. The controllability of various musical aspects, however, has barely been explored. In this paper, we propose Mustango: a music-domain-knowledge-inspired text-to-music system based on diffusion. Mustango aims to control the generated music, not only with general text captions, but with more rich captions that can include specific instructions related to chords, beats, tempo, and key. At the core of Mustango is MuNet, a Music-Domain-Knowledge-Informed UNet guidance module that steers t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.08355","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"eess.AS","submitted_at":"2023-11-14T17:54:38Z","cross_cats_sorted":[],"title_canon_sha256":"378719008f6dba060673275ebcd9dc436edbc8f08777173c049d20cd388041be","abstract_canon_sha256":"59bd04c1c21b5e6697ef7038859dcd71f06c8397185a51dd37c7c2ff15e5d66a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:29.173886Z","signature_b64":"3L9JJBySVVg0tAMlSrbWGfwZeXNqchSRRG7QpN/UFGyI1nACSLsKfjekxGfTKI16mi5rlrPie06kYEG9abQZDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"975db7b957c24046b7471be5a23cf162ef34116bb92ef4fad1d991fc94f6a383","last_reissued_at":"2026-07-05T11:22:29.173387Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:29.173387Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mustango: Toward Controllable Text-to-Music Generation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Deepanway Ghosal, Dorien Herremans, Jan Melechovsky, Navonil Majumder, Soujanya Poria, Zixun Guo","submitted_at":"2023-11-14T17:54:38Z","abstract_excerpt":"The quality of the text-to-music models has reached new heights due to recent advancements in diffusion models. The controllability of various musical aspects, however, has barely been explored. In this paper, we propose Mustango: a music-domain-knowledge-inspired text-to-music system based on diffusion. Mustango aims to control the generated music, not only with general text captions, but with more rich captions that can include specific instructions related to chords, beats, tempo, and key. At the core of Mustango is MuNet, a Music-Domain-Knowledge-Informed UNet guidance module that steers t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.08355","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.08355/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.08355","created_at":"2026-07-05T11:22:29.173448+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.08355v3","created_at":"2026-07-05T11:22:29.173448+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.08355","created_at":"2026-07-05T11:22:29.173448+00:00"},{"alias_kind":"pith_short_12","alias_value":"S5O3POKXYJAE","created_at":"2026-07-05T11:22:29.173448+00:00"},{"alias_kind":"pith_short_16","alias_value":"S5O3POKXYJAENN2H","created_at":"2026-07-05T11:22:29.173448+00:00"},{"alias_kind":"pith_short_8","alias_value":"S5O3POKX","created_at":"2026-07-05T11:22:29.173448+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":80,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06615","citing_title":"FIGMA: Towards FIne-Grained Music retrievAl","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01703","citing_title":"JenBridge: Adaptive Long-Form Video Soundtracking across Scene Transitions","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.19127","citing_title":"Steering Autoregressive Music Generation with Recursive Feature Machines","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2507.08128","citing_title":"Audio Flamingo 3: Advancing Audio Intelligence with Fully Open Large Audio Language Models","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML","json":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML.json","graph_json":"https://pith.science/api/pith-number/S5O3POKXYJAENN2HDPS2EPHRML/graph.json","events_json":"https://pith.science/api/pith-number/S5O3POKXYJAENN2HDPS2EPHRML/events.json","paper":"https://pith.science/paper/S5O3POKX"},"agent_actions":{"view_html":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML","download_json":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML.json","view_paper":"https://pith.science/paper/S5O3POKX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.08355&json=true","fetch_graph":"https://pith.science/api/pith-number/S5O3POKXYJAENN2HDPS2EPHRML/graph.json","fetch_events":"https://pith.science/api/pith-number/S5O3POKXYJAENN2HDPS2EPHRML/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML/action/storage_attestation","attest_author":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML/action/author_attestation","sign_citation":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML/action/citation_signature","submit_replication":"https://pith.science/pith/S5O3POKXYJAENN2HDPS2EPHRML/action/replication_record"}},"created_at":"2026-07-05T11:22:29.173448+00:00","updated_at":"2026-07-05T11:22:29.173448+00:00"}