{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TDVTYJRXJHGSR63EYRYO2L5HNN","short_pith_number":"pith:TDVTYJRX","schema_version":"1.0","canonical_sha256":"98eb3c263749cd28fb64c470ed2fa76b566d679c64a7d5ec8deab36e54d258cf","source":{"kind":"arxiv","id":"2312.09767","version":3},"attestation_state":"computed","paper":{"title":"DreamTalk: When Emotional Talking Head Generation Meets Diffusion Probabilistic Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiayu Wang, Shiwei Zhang, Xiang Wang, Yifeng Ma, Yingya Zhang, Zhidong Deng","submitted_at":"2023-12-15T13:15:42Z","abstract_excerpt":"Emotional talking head generation has attracted growing attention. Previous methods, which are mainly GAN-based, still struggle to consistently produce satisfactory results across diverse emotions and cannot conveniently specify personalized emotions. In this work, we leverage powerful diffusion models to address the issue and propose DreamTalk, a framework that employs meticulous design to unlock the potential of diffusion models in generating emotional talking heads. Specifically, DreamTalk consists of three crucial components: a denoising network, a style-aware lip expert, and a style predi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09767","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-15T13:15:42Z","cross_cats_sorted":[],"title_canon_sha256":"4972cf5ec77deef6b23699e73244d6e257188784cc78e9ddb9329e9c2a891c1f","abstract_canon_sha256":"884641682d731d9f2e8d8d909c1a9c38bdc179c8d186e6765a225f104a32f7f2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:05.824057Z","signature_b64":"jppHpToWGcjreCnfwJKvfnhOwVv7IhF6ZuGuzXokhkqp7qRqv9RMlpOem4WrB0kz3TLT1MpYRhcoa8XPU+SGBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98eb3c263749cd28fb64c470ed2fa76b566d679c64a7d5ec8deab36e54d258cf","last_reissued_at":"2026-07-05T08:54:05.823564Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:05.823564Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DreamTalk: When Emotional Talking Head Generation Meets Diffusion Probabilistic Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiayu Wang, Shiwei Zhang, Xiang Wang, Yifeng Ma, Yingya Zhang, Zhidong Deng","submitted_at":"2023-12-15T13:15:42Z","abstract_excerpt":"Emotional talking head generation has attracted growing attention. Previous methods, which are mainly GAN-based, still struggle to consistently produce satisfactory results across diverse emotions and cannot conveniently specify personalized emotions. In this work, we leverage powerful diffusion models to address the issue and propose DreamTalk, a framework that employs meticulous design to unlock the potential of diffusion models in generating emotional talking heads. Specifically, DreamTalk consists of three crucial components: a denoising network, a style-aware lip expert, and a style predi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09767","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09767/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09767","created_at":"2026-07-05T08:54:05.823622+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09767v3","created_at":"2026-07-05T08:54:05.823622+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09767","created_at":"2026-07-05T08:54:05.823622+00:00"},{"alias_kind":"pith_short_12","alias_value":"TDVTYJRXJHGS","created_at":"2026-07-05T08:54:05.823622+00:00"},{"alias_kind":"pith_short_16","alias_value":"TDVTYJRXJHGSR63E","created_at":"2026-07-05T08:54:05.823622+00:00"},{"alias_kind":"pith_short_8","alias_value":"TDVTYJRX","created_at":"2026-07-05T08:54:05.823622+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01031","citing_title":"Temporally-Aligned Evaluation for Audio-Driven Talking Head Generation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25488","citing_title":"Test-Time Self-Adaptive Conditioning for Stable Audio-Driven Talking-Head Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2411.16748","citing_title":"Multimodal Diffusion Transformer with Memory Bank for Scalable Long-Duration Talking Video Generation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2411.09209","citing_title":"JoyVASA: Portrait and Animal Image Animation with Diffusion-Based Audio-Driven Facial Dynamics and Head Motion Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23552","citing_title":"JAM-Flow: Joint Audio-Motion Synthesis with Flow Matching","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04520","citing_title":"THEval. Evaluation Framework for Talking Head Video Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23632","citing_title":"Hallo-Live: Real-Time Streaming Joint Audio-Video Avatar Generation with Asynchronous Dual-Stream and Human-Centric Preference Distillation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23325","citing_title":"EAD-Net: Emotion-Aware Talking Head Generation with Spatial Refinement and Temporal Coherence","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10367","citing_title":"Beyond Monologue: Interactive Talking-Listening Avatar Generation with Conversational Audio Context-Aware Kernels","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN","json":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN.json","graph_json":"https://pith.science/api/pith-number/TDVTYJRXJHGSR63EYRYO2L5HNN/graph.json","events_json":"https://pith.science/api/pith-number/TDVTYJRXJHGSR63EYRYO2L5HNN/events.json","paper":"https://pith.science/paper/TDVTYJRX"},"agent_actions":{"view_html":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN","download_json":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN.json","view_paper":"https://pith.science/paper/TDVTYJRX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09767&json=true","fetch_graph":"https://pith.science/api/pith-number/TDVTYJRXJHGSR63EYRYO2L5HNN/graph.json","fetch_events":"https://pith.science/api/pith-number/TDVTYJRXJHGSR63EYRYO2L5HNN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN/action/storage_attestation","attest_author":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN/action/author_attestation","sign_citation":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN/action/citation_signature","submit_replication":"https://pith.science/pith/TDVTYJRXJHGSR63EYRYO2L5HNN/action/replication_record"}},"created_at":"2026-07-05T08:54:05.823622+00:00","updated_at":"2026-07-05T08:54:05.823622+00:00"}