{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GHA2MR6OGLDXESPQW4JCKGL7ES","short_pith_number":"pith:GHA2MR6O","schema_version":"1.0","canonical_sha256":"31c1a647ce32c77249f0b71225197f2489b627b58a2e11401178940f0c315733","source":{"kind":"arxiv","id":"2505.22647","version":1},"attestation_state":"computed","paper":{"title":"Let Them Talk: Audio-Driven Multi-Person Conversational Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Gao, Guanying Chen, Wenhan Luo, Xiaoming Wei, Xunliang Cai, Yong Zhang, Zhe Kong, Zhuoliang Kang","submitted_at":"2025-05-28T17:57:06Z","abstract_excerpt":"Audio-driven human animation methods, such as talking head and talking body generation, have made remarkable progress in generating synchronized facial movements and appealing visual quality videos. However, existing methods primarily focus on single human animation and struggle with multi-stream audio inputs, facing incorrect binding problems between audio and persons. Additionally, they exhibit limitations in instruction-following capabilities. To solve this problem, in this paper, we propose a novel task: Multi-Person Conversational Video Generation, and introduce a new framework, MultiTalk"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22647","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-28T17:57:06Z","cross_cats_sorted":[],"title_canon_sha256":"9e72678420a0fc40527f825abba8996bd16ceb13ae5bfd760b76c77ea82d9926","abstract_canon_sha256":"309d52a4dd3609bf71212df849cbcb570141f3a887bc7d8aa8941405afe1b670"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:28.430169Z","signature_b64":"0gmIEBVmaemnzbf6LVbDsjxXuaHleP5ixl+tq7+wCyd9j7+RyeW+otvfKi8x9mhcqM20ic5hsJIsOLlamaxLAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31c1a647ce32c77249f0b71225197f2489b627b58a2e11401178940f0c315733","last_reissued_at":"2026-07-05T11:11:28.429621Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:28.429621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Let Them Talk: Audio-Driven Multi-Person Conversational Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Gao, Guanying Chen, Wenhan Luo, Xiaoming Wei, Xunliang Cai, Yong Zhang, Zhe Kong, Zhuoliang Kang","submitted_at":"2025-05-28T17:57:06Z","abstract_excerpt":"Audio-driven human animation methods, such as talking head and talking body generation, have made remarkable progress in generating synchronized facial movements and appealing visual quality videos. However, existing methods primarily focus on single human animation and struggle with multi-stream audio inputs, facing incorrect binding problems between audio and persons. Additionally, they exhibit limitations in instruction-following capabilities. To solve this problem, in this paper, we propose a novel task: Multi-Person Conversational Video Generation, and introduce a new framework, MultiTalk"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22647","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22647","created_at":"2026-07-05T11:11:28.429681+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22647v1","created_at":"2026-07-05T11:11:28.429681+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22647","created_at":"2026-07-05T11:11:28.429681+00:00"},{"alias_kind":"pith_short_12","alias_value":"GHA2MR6OGLDX","created_at":"2026-07-05T11:11:28.429681+00:00"},{"alias_kind":"pith_short_16","alias_value":"GHA2MR6OGLDXESPQ","created_at":"2026-07-05T11:11:28.429681+00:00"},{"alias_kind":"pith_short_8","alias_value":"GHA2MR6O","created_at":"2026-07-05T11:11:28.429681+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22905","citing_title":"InteractiveAvatar: Real-Time Streaming Video Generation for Consistent and Intent-Aware Avatars","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31088","citing_title":"Towards Flexible, Natural, Efficient Interaction for Conversational Talking Face Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22905","citing_title":"InteractiveAvatar: Real-Time Streaming Video Generation for Consistent and Intent-Aware Avatars","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25488","citing_title":"Test-Time Self-Adaptive Conditioning for Stable Audio-Driven Talking-Head Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09534","citing_title":"AUHead: Realistic Emotional Talking Head Generation via Action Units Control","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11363","citing_title":"PresentAgent-2: Towards Generalist Multimodal Presentation Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11804","citing_title":"OmniShow: Unifying Multimodal Conditions for Human-Object Interaction Video Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14580","citing_title":"TurboTalk: Progressive Distillation for One-Step Audio-Driven Talking Avatar Generation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES","json":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES.json","graph_json":"https://pith.science/api/pith-number/GHA2MR6OGLDXESPQW4JCKGL7ES/graph.json","events_json":"https://pith.science/api/pith-number/GHA2MR6OGLDXESPQW4JCKGL7ES/events.json","paper":"https://pith.science/paper/GHA2MR6O"},"agent_actions":{"view_html":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES","download_json":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES.json","view_paper":"https://pith.science/paper/GHA2MR6O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22647&json=true","fetch_graph":"https://pith.science/api/pith-number/GHA2MR6OGLDXESPQW4JCKGL7ES/graph.json","fetch_events":"https://pith.science/api/pith-number/GHA2MR6OGLDXESPQW4JCKGL7ES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES/action/storage_attestation","attest_author":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES/action/author_attestation","sign_citation":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES/action/citation_signature","submit_replication":"https://pith.science/pith/GHA2MR6OGLDXESPQW4JCKGL7ES/action/replication_record"}},"created_at":"2026-07-05T11:11:28.429681+00:00","updated_at":"2026-07-05T11:11:28.429681+00:00"}