{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5RQGVECUR4S5CHD5M5MN3AVJTG","short_pith_number":"pith:5RQGVECU","schema_version":"1.0","canonical_sha256":"ec606a90548f25d11c7d6758dd82a9998ee14cc8eab3bc0f2b01c0f404eea78f","source":{"kind":"arxiv","id":"2403.09407","version":1},"attestation_state":"computed","paper":{"title":"LM2D: Lyrics- and Music-Driven Dance Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Danica Kragic, Hang Yin, M{\\aa}rten Bj\\\"orkman, Wenjie Yin, Xuejiao Zhao, Yi Yu","submitted_at":"2024-03-14T13:59:04Z","abstract_excerpt":"Dance typically involves professional choreography with complex movements that follow a musical rhythm and can also be influenced by lyrical content. The integration of lyrics in addition to the auditory dimension, enriches the foundational tone and makes motion generation more amenable to its semantic meanings. However, existing dance synthesis methods tend to model motions only conditioned on audio signals. In this work, we make two contributions to bridge this gap. First, we propose LM2D, a novel probabilistic architecture that incorporates a multimodal diffusion model with consistency dist"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.09407","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SD","submitted_at":"2024-03-14T13:59:04Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM","eess.AS"],"title_canon_sha256":"ab19167227a81b6928205d0dd36ea418a7338d78067a2afb11b2afba18e1bd3a","abstract_canon_sha256":"95b73f9847dfbdb2873cb60043dbe5b2eddea07d64042da21343618b7caa1536"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:09.790958Z","signature_b64":"4uvlBwBGhd0OmFvc5j5U+ikz8103RzM6P9G9rGEbyGpeYfLkUNmlqTQJVCe18UEORT6fAE9V50aAli+xc6ZlAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec606a90548f25d11c7d6758dd82a9998ee14cc8eab3bc0f2b01c0f404eea78f","last_reissued_at":"2026-07-05T07:56:09.790543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:09.790543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LM2D: Lyrics- and Music-Driven Dance Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","eess.AS"],"primary_cat":"cs.SD","authors_text":"Danica Kragic, Hang Yin, M{\\aa}rten Bj\\\"orkman, Wenjie Yin, Xuejiao Zhao, Yi Yu","submitted_at":"2024-03-14T13:59:04Z","abstract_excerpt":"Dance typically involves professional choreography with complex movements that follow a musical rhythm and can also be influenced by lyrical content. The integration of lyrics in addition to the auditory dimension, enriches the foundational tone and makes motion generation more amenable to its semantic meanings. However, existing dance synthesis methods tend to model motions only conditioned on audio signals. In this work, we make two contributions to bridge this gap. First, we propose LM2D, a novel probabilistic architecture that incorporates a multimodal diffusion model with consistency dist"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.09407","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.09407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.09407","created_at":"2026-07-05T07:56:09.790603+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.09407v1","created_at":"2026-07-05T07:56:09.790603+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.09407","created_at":"2026-07-05T07:56:09.790603+00:00"},{"alias_kind":"pith_short_12","alias_value":"5RQGVECUR4S5","created_at":"2026-07-05T07:56:09.790603+00:00"},{"alias_kind":"pith_short_16","alias_value":"5RQGVECUR4S5CHD5","created_at":"2026-07-05T07:56:09.790603+00:00"},{"alias_kind":"pith_short_8","alias_value":"5RQGVECU","created_at":"2026-07-05T07:56:09.790603+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.03323","citing_title":"Listen to Rhythm, Choose Movements: Autoregressive Multimodal Dance Generation via Diffusion and Mamba with Decoupled Dance Dataset","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14525","citing_title":"From Sparse to Dense: Spatio-Temporal Fusion for Multi-View 3D Human Pose Estimation with DenseWarper","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":166,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG","json":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG.json","graph_json":"https://pith.science/api/pith-number/5RQGVECUR4S5CHD5M5MN3AVJTG/graph.json","events_json":"https://pith.science/api/pith-number/5RQGVECUR4S5CHD5M5MN3AVJTG/events.json","paper":"https://pith.science/paper/5RQGVECU"},"agent_actions":{"view_html":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG","download_json":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG.json","view_paper":"https://pith.science/paper/5RQGVECU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.09407&json=true","fetch_graph":"https://pith.science/api/pith-number/5RQGVECUR4S5CHD5M5MN3AVJTG/graph.json","fetch_events":"https://pith.science/api/pith-number/5RQGVECUR4S5CHD5M5MN3AVJTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG/action/storage_attestation","attest_author":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG/action/author_attestation","sign_citation":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG/action/citation_signature","submit_replication":"https://pith.science/pith/5RQGVECUR4S5CHD5M5MN3AVJTG/action/replication_record"}},"created_at":"2026-07-05T07:56:09.790603+00:00","updated_at":"2026-07-05T07:56:09.790603+00:00"}