{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3EJC6VS2YS2CCXSQWUUW3I7VIG","short_pith_number":"pith:3EJC6VS2","schema_version":"1.0","canonical_sha256":"d9122f565ac4b4215e50b5296da3f541987b060a6b410977858ae6058b7a940c","source":{"kind":"arxiv","id":"2503.19611","version":1},"attestation_state":"computed","paper":{"title":"Analyzable Chain-of-Musical-Thought Prompting for High-Fidelity Music Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.MM","eess.AS","eess.SP"],"primary_cat":"cs.SD","authors_text":"Chien-Hung Liu, Feng Liu, Fuqiang Jiang, Hangyu Liu, Hanyu Chen, Jingcheng Wu, Max W. Y. Lam, Tianwei Zhao, Tong Feng, Wei-Tsung Lu, Weiya You, Xingda Li, Xuchen Song, Yahui Zhou, Yang Li, Yijin Xing, Zongyu Yin","submitted_at":"2025-03-25T12:51:21Z","abstract_excerpt":"Autoregressive (AR) models have demonstrated impressive capabilities in generating high-fidelity music. However, the conventional next-token prediction paradigm in AR models does not align with the human creative process in music composition, potentially compromising the musicality of generated samples. To overcome this limitation, we introduce MusiCoT, a novel chain-of-thought (CoT) prompting technique tailored for music generation. MusiCoT empowers the AR model to first outline an overall music structure before generating audio tokens, thereby enhancing the coherence and creativity of the re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.19611","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SD","submitted_at":"2025-03-25T12:51:21Z","cross_cats_sorted":["cs.AI","cs.MM","eess.AS","eess.SP"],"title_canon_sha256":"786b6618a1784f2f12d84ba6c424002b6c4edcf5ab43db6e914fd79bc0afde43","abstract_canon_sha256":"417ad5f75869fcbabbc5fa278486367df52f011c0a0d14572bd1a1bdd1653949"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:58.372894Z","signature_b64":"Dvfn4xeBwKVg6TsUcPicLYek266BNCrFBFUpAPL+YdqMXYTPhJaQ5Ry+WfcWbtzFdQytNeTtKkm7cYa50l5lAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d9122f565ac4b4215e50b5296da3f541987b060a6b410977858ae6058b7a940c","last_reissued_at":"2026-07-05T10:38:58.372352Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:58.372352Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzable Chain-of-Musical-Thought Prompting for High-Fidelity Music Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.MM","eess.AS","eess.SP"],"primary_cat":"cs.SD","authors_text":"Chien-Hung Liu, Feng Liu, Fuqiang Jiang, Hangyu Liu, Hanyu Chen, Jingcheng Wu, Max W. Y. Lam, Tianwei Zhao, Tong Feng, Wei-Tsung Lu, Weiya You, Xingda Li, Xuchen Song, Yahui Zhou, Yang Li, Yijin Xing, Zongyu Yin","submitted_at":"2025-03-25T12:51:21Z","abstract_excerpt":"Autoregressive (AR) models have demonstrated impressive capabilities in generating high-fidelity music. However, the conventional next-token prediction paradigm in AR models does not align with the human creative process in music composition, potentially compromising the musicality of generated samples. To overcome this limitation, we introduce MusiCoT, a novel chain-of-thought (CoT) prompting technique tailored for music generation. MusiCoT empowers the AR model to first outline an overall music structure before generating audio tokens, thereby enhancing the coherence and creativity of the re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.19611","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.19611/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.19611","created_at":"2026-07-05T10:38:58.372415+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.19611v1","created_at":"2026-07-05T10:38:58.372415+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.19611","created_at":"2026-07-05T10:38:58.372415+00:00"},{"alias_kind":"pith_short_12","alias_value":"3EJC6VS2YS2C","created_at":"2026-07-05T10:38:58.372415+00:00"},{"alias_kind":"pith_short_16","alias_value":"3EJC6VS2YS2CCXSQ","created_at":"2026-07-05T10:38:58.372415+00:00"},{"alias_kind":"pith_short_8","alias_value":"3EJC6VS2","created_at":"2026-07-05T10:38:58.372415+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":116,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21227","citing_title":"Bagpiper-Edit: Zero-Shot Open-Ended Audio Editing via Rich-Caption","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03169","citing_title":"SketchSong: Hierarchical Song Generation with Sketch Planning and Fine-Grained Multi-Track Modeling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01703","citing_title":"JenBridge: Adaptive Long-Form Video Soundtracking across Scene Transitions","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01677","citing_title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30642","citing_title":"LeVo 2: Stable and Melodious Song Generation via Hierarchical Representation Modeling and Progressive Post-Training","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG","json":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG.json","graph_json":"https://pith.science/api/pith-number/3EJC6VS2YS2CCXSQWUUW3I7VIG/graph.json","events_json":"https://pith.science/api/pith-number/3EJC6VS2YS2CCXSQWUUW3I7VIG/events.json","paper":"https://pith.science/paper/3EJC6VS2"},"agent_actions":{"view_html":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG","download_json":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG.json","view_paper":"https://pith.science/paper/3EJC6VS2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.19611&json=true","fetch_graph":"https://pith.science/api/pith-number/3EJC6VS2YS2CCXSQWUUW3I7VIG/graph.json","fetch_events":"https://pith.science/api/pith-number/3EJC6VS2YS2CCXSQWUUW3I7VIG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG/action/storage_attestation","attest_author":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG/action/author_attestation","sign_citation":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG/action/citation_signature","submit_replication":"https://pith.science/pith/3EJC6VS2YS2CCXSQWUUW3I7VIG/action/replication_record"}},"created_at":"2026-07-05T10:38:58.372415+00:00","updated_at":"2026-07-05T10:38:58.372415+00:00"}