{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LZNRPQYRP7YAZFFHDZKCCG4WFG","short_pith_number":"pith:LZNRPQYR","schema_version":"1.0","canonical_sha256":"5e5b17c3117ff00c94a71e54211b962990910df5ba124c8255234dc34589f5db","source":{"kind":"arxiv","id":"2501.18801","version":1},"attestation_state":"computed","paper":{"title":"Every Image Listens, Every Image Dances: Music-Driven Image Animation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ju-Chiang Wang, Pawel Polak, Peng Zhang, Weituo Hao, Zhikang Dong","submitted_at":"2025-01-30T23:38:51Z","abstract_excerpt":"Image animation has become a promising area in multimodal research, with a focus on generating videos from reference images. While prior work has largely emphasized generic video generation guided by text, music-driven dance video generation remains underexplored. In this paper, we introduce MuseDance, an innovative end-to-end model that animates reference images using both music and text inputs. This dual input enables MuseDance to generate personalized videos that follow text descriptions and synchronize character movements with the music. Unlike existing approaches, MuseDance eliminates the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.18801","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-30T23:38:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1a52f5acce1ee58c5cb000ba94df2b07aeacce18ec04270723583a098381ab1e","abstract_canon_sha256":"b608dc1b79ed2cd8f601052de479a987abed697d71db8e725a504453d9218ade"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:51.809192Z","signature_b64":"HP65n/8OwSkmDl/aCNmWA/fLRvzqZOtBh+dFNzLGudc/W9T07ymvj4TTQO6WgQiVAJi8/ev+cV2pDNuLYtNlBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e5b17c3117ff00c94a71e54211b962990910df5ba124c8255234dc34589f5db","last_reissued_at":"2026-07-05T10:07:51.808722Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:51.808722Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Every Image Listens, Every Image Dances: Music-Driven Image Animation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ju-Chiang Wang, Pawel Polak, Peng Zhang, Weituo Hao, Zhikang Dong","submitted_at":"2025-01-30T23:38:51Z","abstract_excerpt":"Image animation has become a promising area in multimodal research, with a focus on generating videos from reference images. While prior work has largely emphasized generic video generation guided by text, music-driven dance video generation remains underexplored. In this paper, we introduce MuseDance, an innovative end-to-end model that animates reference images using both music and text inputs. This dual input enables MuseDance to generate personalized videos that follow text descriptions and synchronize character movements with the music. Unlike existing approaches, MuseDance eliminates the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.18801","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.18801/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.18801","created_at":"2026-07-05T10:07:51.808776+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.18801v1","created_at":"2026-07-05T10:07:51.808776+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.18801","created_at":"2026-07-05T10:07:51.808776+00:00"},{"alias_kind":"pith_short_12","alias_value":"LZNRPQYRP7YA","created_at":"2026-07-05T10:07:51.808776+00:00"},{"alias_kind":"pith_short_16","alias_value":"LZNRPQYRP7YAZFFH","created_at":"2026-07-05T10:07:51.808776+00:00"},{"alias_kind":"pith_short_8","alias_value":"LZNRPQYR","created_at":"2026-07-05T10:07:51.808776+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30019","citing_title":"OmniDance: Multimodal Driven Dance Video Generation with Large-scale Internet Data","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG","json":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG.json","graph_json":"https://pith.science/api/pith-number/LZNRPQYRP7YAZFFHDZKCCG4WFG/graph.json","events_json":"https://pith.science/api/pith-number/LZNRPQYRP7YAZFFHDZKCCG4WFG/events.json","paper":"https://pith.science/paper/LZNRPQYR"},"agent_actions":{"view_html":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG","download_json":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG.json","view_paper":"https://pith.science/paper/LZNRPQYR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.18801&json=true","fetch_graph":"https://pith.science/api/pith-number/LZNRPQYRP7YAZFFHDZKCCG4WFG/graph.json","fetch_events":"https://pith.science/api/pith-number/LZNRPQYRP7YAZFFHDZKCCG4WFG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG/action/storage_attestation","attest_author":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG/action/author_attestation","sign_citation":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG/action/citation_signature","submit_replication":"https://pith.science/pith/LZNRPQYRP7YAZFFHDZKCCG4WFG/action/replication_record"}},"created_at":"2026-07-05T10:07:51.808776+00:00","updated_at":"2026-07-05T10:07:51.808776+00:00"}