{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CTYJPZXSO3NFWM6MW2ZOESGKCI","short_pith_number":"pith:CTYJPZXS","schema_version":"1.0","canonical_sha256":"14f097e6f276da5b33ccb6b2e248ca1205704cbc6ee62d06e71c870bd37eb513","source":{"kind":"arxiv","id":"2511.03942","version":2},"attestation_state":"computed","paper":{"title":"MIDI-LLM: Improving Text-to-MIDI Music Generation via Adapting Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.SD","authors_text":"Cheng-Zhi Anna Huang, Chris Donahue, Dave Carlton, Ryan Miyakawa, Shih-Lun Wu, Yoon Kim","submitted_at":"2025-11-06T00:40:07Z","abstract_excerpt":"We present MIDI-LLM, a recipe that improves multitrack text-to-MIDI generation via adapting Large Language Models (LLMs). MIDI-LLM expands an LLM's text vocabulary to include MIDI tokens and employs a two-stage training pipeline: (i) unimodal continued pretraining on music-adjacent text and standalone MIDIs, and (ii) multimodal supervised finetuning on text-MIDI pairs. Our instantiation of MIDI-LLM based on Llama 3.2 (1B) outperforms the recent Text2midi model in both text control and musical quality, and readily integrates with optimized inference ecosystems like vLLM. To align with real-worl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2511.03942","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2025-11-06T00:40:07Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"d2efa23ce5b99dc2a4c887b3fb38df86cf3a19de2abdc3583084c2b14bcac061","abstract_canon_sha256":"9e971bb886d6b270f17b608a4f1582be5d18529ad5103e07c7a3d02cdcfa9e12"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-05T00:41:00.198371Z","signature_b64":"OPH3iH94XD3tzYXczF2TkRGTAcNc9pWwc4ORfgWbw+ETQIU2N5zwRtQV4XxSTv+NWLAF3Et61jU1HlOvOpccCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14f097e6f276da5b33ccb6b2e248ca1205704cbc6ee62d06e71c870bd37eb513","last_reissued_at":"2026-08-05T00:41:00.194700Z","signature_status":"signed_v1","first_computed_at":"2026-08-05T00:41:00.194700Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MIDI-LLM: Improving Text-to-MIDI Music Generation via Adapting Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.SD","authors_text":"Cheng-Zhi Anna Huang, Chris Donahue, Dave Carlton, Ryan Miyakawa, Shih-Lun Wu, Yoon Kim","submitted_at":"2025-11-06T00:40:07Z","abstract_excerpt":"We present MIDI-LLM, a recipe that improves multitrack text-to-MIDI generation via adapting Large Language Models (LLMs). MIDI-LLM expands an LLM's text vocabulary to include MIDI tokens and employs a two-stage training pipeline: (i) unimodal continued pretraining on music-adjacent text and standalone MIDIs, and (ii) multimodal supervised finetuning on text-MIDI pairs. Our instantiation of MIDI-LLM based on Llama 3.2 (1B) outperforms the recent Text2midi model in both text control and musical quality, and readily integrates with optimized inference ecosystems like vLLM. To align with real-worl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2511.03942","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2511.03942/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2511.03942","created_at":"2026-08-05T00:41:00.196017+00:00"},{"alias_kind":"arxiv_version","alias_value":"2511.03942v2","created_at":"2026-08-05T00:41:00.196017+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2511.03942","created_at":"2026-08-05T00:41:00.196017+00:00"},{"alias_kind":"pith_short_12","alias_value":"CTYJPZXSO3NF","created_at":"2026-08-05T00:41:00.196017+00:00"},{"alias_kind":"pith_short_16","alias_value":"CTYJPZXSO3NFWM6M","created_at":"2026-08-05T00:41:00.196017+00:00"},{"alias_kind":"pith_short_8","alias_value":"CTYJPZXS","created_at":"2026-08-05T00:41:00.196017+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2605.13431","citing_title":"Text2Score: Generating Sheet Music From Textual Prompts","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2604.19532","citing_title":"BEAT: Tokenizing and Generating Symbolic Music by Uniform Temporal Steps","ref_index":50,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI","json":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI.json","graph_json":"https://pith.science/api/pith-number/CTYJPZXSO3NFWM6MW2ZOESGKCI/graph.json","events_json":"https://pith.science/api/pith-number/CTYJPZXSO3NFWM6MW2ZOESGKCI/events.json","paper":"https://pith.science/paper/CTYJPZXS"},"agent_actions":{"view_html":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI","download_json":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI.json","view_paper":"https://pith.science/paper/CTYJPZXS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2511.03942&json=true","fetch_graph":"https://pith.science/api/pith-number/CTYJPZXSO3NFWM6MW2ZOESGKCI/graph.json","fetch_events":"https://pith.science/api/pith-number/CTYJPZXSO3NFWM6MW2ZOESGKCI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI/action/storage_attestation","attest_author":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI/action/author_attestation","sign_citation":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI/action/citation_signature","submit_replication":"https://pith.science/pith/CTYJPZXSO3NFWM6MW2ZOESGKCI/action/replication_record"}},"created_at":"2026-08-05T00:41:00.196017+00:00","updated_at":"2026-08-05T00:41:00.196017+00:00"}