{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6G5SCEMCA3DOHYTKZL7DT2CURZ","short_pith_number":"pith:6G5SCEMC","schema_version":"1.0","canonical_sha256":"f1bb21118206c6e3e26acafe39e8548e62d50d510c88a426cda25a614635138f","source":{"kind":"arxiv","id":"2503.01183","version":1},"attestation_state":"computed","paper":{"title":"DiffRhythm: Blazingly Fast and Embarrassingly Simple End-to-End Full-Length Song Generation with Latent Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Chunbo Hao, Guobin Ma, Huakang Chen, Jixun Yao, Lei Xie, Shuai Wang, Yuepeng Jiang, Ziqian Ning","submitted_at":"2025-03-03T05:15:34Z","abstract_excerpt":"Recent advancements in music generation have garnered significant attention, yet existing approaches face critical limitations. Some current generative models can only synthesize either the vocal track or the accompaniment track. While some models can generate combined vocal and accompaniment, they typically rely on meticulously designed multi-stage cascading architectures and intricate data pipelines, hindering scalability. Additionally, most systems are restricted to generating short musical segments rather than full-length songs. Furthermore, widely used language model-based methods suffer "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01183","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-03-03T05:15:34Z","cross_cats_sorted":[],"title_canon_sha256":"e1d7d6bc3e0ee80001d5b1ef38007de177aa50199fd76a50e484c3677d526d09","abstract_canon_sha256":"ffed64a66c073a043d5295e0a1c5cc437c000eaab980ca6af5a3b09240bd07a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:01.047227Z","signature_b64":"aE0QsVIVqp+2E8LE91fYnAQ1o0eO3ac+Ugw2G/M+YZzc8TCbHCCILdCSEuvbJ+pC+EhGPYnlzC2lFtz9eptOAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1bb21118206c6e3e26acafe39e8548e62d50d510c88a426cda25a614635138f","last_reissued_at":"2026-07-05T10:23:01.046707Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:01.046707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiffRhythm: Blazingly Fast and Embarrassingly Simple End-to-End Full-Length Song Generation with Latent Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Chunbo Hao, Guobin Ma, Huakang Chen, Jixun Yao, Lei Xie, Shuai Wang, Yuepeng Jiang, Ziqian Ning","submitted_at":"2025-03-03T05:15:34Z","abstract_excerpt":"Recent advancements in music generation have garnered significant attention, yet existing approaches face critical limitations. Some current generative models can only synthesize either the vocal track or the accompaniment track. While some models can generate combined vocal and accompaniment, they typically rely on meticulously designed multi-stage cascading architectures and intricate data pipelines, hindering scalability. Additionally, most systems are restricted to generating short musical segments rather than full-length songs. Furthermore, widely used language model-based methods suffer "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01183","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01183","created_at":"2026-07-05T10:23:01.046770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01183v1","created_at":"2026-07-05T10:23:01.046770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01183","created_at":"2026-07-05T10:23:01.046770+00:00"},{"alias_kind":"pith_short_12","alias_value":"6G5SCEMCA3DO","created_at":"2026-07-05T10:23:01.046770+00:00"},{"alias_kind":"pith_short_16","alias_value":"6G5SCEMCA3DOHYTK","created_at":"2026-07-05T10:23:01.046770+00:00"},{"alias_kind":"pith_short_8","alias_value":"6G5SCEMC","created_at":"2026-07-05T10:23:01.046770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06986","citing_title":"MMGenre: Benchmarking Singing Voice Synthesis across Multiple Musical Genres","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18985","citing_title":"SingFox: A Multi-Lingual Singfake Detection Corpus","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18052","citing_title":"An Empirical Analysis of AI Slop in Music Streaming","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08663","citing_title":"Probing Token Spaces under Generator Shift in AI-Generated Music Detection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07015","citing_title":"Towards Unified Song Generation and Singing Voice Conversion with Accompaniment Co-Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03169","citing_title":"SketchSong: Hierarchical Song Generation with Sketch Planning and Fine-Grained Multi-Track Modeling","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30642","citing_title":"LeVo 2: Stable and Melodious Song Generation via Hierarchical Representation Modeling and Progressive Post-Training","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17414","citing_title":"S2Accompanist: A Semantic-Aware and Structure-Guided Diffusion Model for Music Accompaniment Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02797","citing_title":"SongFormer: Scaling Music Structure Analysis with Heterogeneous Supervision","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22029","citing_title":"MIDI-Informed Singing Accompaniment Generation in a Compositional Song Pipeline","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2603.24589","citing_title":"YingMusic-Singer: Controllable Singing Voice Synthesis with Flexible Lyric Manipulation and Annotation-free Melody Guidance","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25937","citing_title":"SongBench: A Fine-Grained Multi-Aspect Benchmark for Song Quality Assessment","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ","json":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ.json","graph_json":"https://pith.science/api/pith-number/6G5SCEMCA3DOHYTKZL7DT2CURZ/graph.json","events_json":"https://pith.science/api/pith-number/6G5SCEMCA3DOHYTKZL7DT2CURZ/events.json","paper":"https://pith.science/paper/6G5SCEMC"},"agent_actions":{"view_html":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ","download_json":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ.json","view_paper":"https://pith.science/paper/6G5SCEMC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01183&json=true","fetch_graph":"https://pith.science/api/pith-number/6G5SCEMCA3DOHYTKZL7DT2CURZ/graph.json","fetch_events":"https://pith.science/api/pith-number/6G5SCEMCA3DOHYTKZL7DT2CURZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ/action/storage_attestation","attest_author":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ/action/author_attestation","sign_citation":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ/action/citation_signature","submit_replication":"https://pith.science/pith/6G5SCEMCA3DOHYTKZL7DT2CURZ/action/replication_record"}},"created_at":"2026-07-05T10:23:01.046770+00:00","updated_at":"2026-07-05T10:23:01.046770+00:00"}