{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6MPZ4J6D72CKSBDU7EXDAI627W","short_pith_number":"pith:6MPZ4J6D","schema_version":"1.0","canonical_sha256":"f31f9e27c3fe84a90474f92e3023dafd8f3d39079873b75dccee6f77e0d4515a","source":{"kind":"arxiv","id":"2508.18621","version":1},"attestation_state":"computed","paper":{"title":"Wan-S2V: Audio-Driven Cinematic Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bang Zhang, Chaonan Ji, Dechao Meng, Guangyuan Wang, Jiayu Xiao, Jingren Zhou, Jinwei Qi, Ke Sun, Lian Zhuo, Li Hu, Linrui Tian, Mingyang Huang, Penchong Qiao, Peng Zhang, Qi Wang, Sheng Xu, Siqi Hu, Xindi Zhang, Xin Gao, Yafei Song, Zhen Shen, Zhe Zhang, Zhongjian Wang","submitted_at":"2025-08-26T02:51:31Z","abstract_excerpt":"Current state-of-the-art (SOTA) methods for audio-driven character animation demonstrate promising performance for scenarios primarily involving speech and singing. However, they often fall short in more complex film and television productions, which demand sophisticated elements such as nuanced character interactions, realistic body movements, and dynamic camera work. To address this long-standing challenge of achieving film-level character animation, we propose an audio-driven model, which we refere to as Wan-S2V, built upon Wan. Our model achieves significantly enhanced expressiveness and f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.18621","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-26T02:51:31Z","cross_cats_sorted":[],"title_canon_sha256":"7a5152b9c95d04f24d75756ed8b15b5710a3540cc127fdbfe8cf6ed5c3e72499","abstract_canon_sha256":"bde9c7f777815dda6818622e743e498f60cd61cabadab936ed89e607b3f7c5f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:27.990731Z","signature_b64":"n09ogSdEtqLtZSwhclTKIoB4IOk1weiodANpGwdw2tch/kU2PE8/FQbWuW8jq7nqyi3zFD0rV4BygDyGZkIXCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f31f9e27c3fe84a90474f92e3023dafd8f3d39079873b75dccee6f77e0d4515a","last_reissued_at":"2026-07-05T11:59:27.990278Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:27.990278Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Wan-S2V: Audio-Driven Cinematic Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bang Zhang, Chaonan Ji, Dechao Meng, Guangyuan Wang, Jiayu Xiao, Jingren Zhou, Jinwei Qi, Ke Sun, Lian Zhuo, Li Hu, Linrui Tian, Mingyang Huang, Penchong Qiao, Peng Zhang, Qi Wang, Sheng Xu, Siqi Hu, Xindi Zhang, Xin Gao, Yafei Song, Zhen Shen, Zhe Zhang, Zhongjian Wang","submitted_at":"2025-08-26T02:51:31Z","abstract_excerpt":"Current state-of-the-art (SOTA) methods for audio-driven character animation demonstrate promising performance for scenarios primarily involving speech and singing. However, they often fall short in more complex film and television productions, which demand sophisticated elements such as nuanced character interactions, realistic body movements, and dynamic camera work. To address this long-standing challenge of achieving film-level character animation, we propose an audio-driven model, which we refere to as Wan-S2V, built upon Wan. Our model achieves significantly enhanced expressiveness and f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.18621","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.18621/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.18621","created_at":"2026-07-05T11:59:27.990339+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.18621v1","created_at":"2026-07-05T11:59:27.990339+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.18621","created_at":"2026-07-05T11:59:27.990339+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MPZ4J6D72CK","created_at":"2026-07-05T11:59:27.990339+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MPZ4J6D72CKSBDU","created_at":"2026-07-05T11:59:27.990339+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MPZ4J6D","created_at":"2026-07-05T11:59:27.990339+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13304","citing_title":"ReFree: Towards Realistic Co-Speech Video Generation via Reward-Free RL and Multilevel Speech Guidance","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01222","citing_title":"Ink3D: Sculpting 3D Assets with Extremely Complex Textures via Video Generative Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30849","citing_title":"SyncCache: Exploiting Asymmetric Dynamics for Fast Audio-Driven Portrait Animation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30019","citing_title":"OmniDance: Multimodal Driven Dance Video Generation with Large-scale Internet Data","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25659","citing_title":"StreamChar: Long-Horizon Streaming Character Audio-Video Generation with Decoupled Orchestration","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26486","citing_title":"LongCat-Video-Avatar 1.5 Technical Report","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17488","citing_title":"Omni-Customizer: End-to-End MultiModal Customization for Joint Audio-Video Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01186","citing_title":"ASTRA: Let Arbitrary Subjects Transform in Video Editing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04677","citing_title":"Live Avatar: Streaming Real-time Audio-Driven Avatar Generation with Infinite Length","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09534","citing_title":"AUHead: Realistic Emotional Talking Head Generation via Action Units Control","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13669","citing_title":"EchoTorrent: Towards Swift, Sustained, and Streaming Multi-Modal Video Generation","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03233","citing_title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11424","citing_title":"VidSplat: Gaussian Splatting Reconstruction with Geometry-Guided Video Diffusion Priors","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27918","citing_title":"Generate Your Talking Avatar from Video Reference","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25819","citing_title":"Mutual Forcing: Dual-Mode Self-Evolution for Fast Autoregressive Audio-Video Character Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06192","citing_title":"EA-WM: Event-Aware Generative World Model with Structured Kinematic-to-Visual Action Fields","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19636","citing_title":"CoInteract: Physically-Consistent Human-Object Interaction Video Synthesis via Spatially-Structured Co-Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07823","citing_title":"LPM 1.0: Video-based Character Performance Model","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09057","citing_title":"Tora3: Trajectory-Guided Audio-Video Generation with Physical Coherence","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W","json":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W.json","graph_json":"https://pith.science/api/pith-number/6MPZ4J6D72CKSBDU7EXDAI627W/graph.json","events_json":"https://pith.science/api/pith-number/6MPZ4J6D72CKSBDU7EXDAI627W/events.json","paper":"https://pith.science/paper/6MPZ4J6D"},"agent_actions":{"view_html":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W","download_json":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W.json","view_paper":"https://pith.science/paper/6MPZ4J6D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.18621&json=true","fetch_graph":"https://pith.science/api/pith-number/6MPZ4J6D72CKSBDU7EXDAI627W/graph.json","fetch_events":"https://pith.science/api/pith-number/6MPZ4J6D72CKSBDU7EXDAI627W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W/action/storage_attestation","attest_author":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W/action/author_attestation","sign_citation":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W/action/citation_signature","submit_replication":"https://pith.science/pith/6MPZ4J6D72CKSBDU7EXDAI627W/action/replication_record"}},"created_at":"2026-07-05T11:59:27.990339+00:00","updated_at":"2026-07-05T11:59:27.990339+00:00"}