{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:V6IQDGHYWSLRS65CQJFMA36RNV","short_pith_number":"pith:V6IQDGHY","schema_version":"1.0","canonical_sha256":"af910198f8b497197ba2824ac06fd16d720093d4324318eb58264ceadb199bcb","source":{"kind":"arxiv","id":"2301.03786","version":2},"attestation_state":"computed","paper":{"title":"DiffTalk: Crafting Diffusion Models for Generalized Audio-Driven Portraits Animation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jie Zhou, Jiwen Lu, Shuai Shen, Wanhua Li, Wenliang Zhao, Zheng Zhu, Zibin Meng","submitted_at":"2023-01-10T05:11:25Z","abstract_excerpt":"Talking head synthesis is a promising approach for the video production industry. Recently, a lot of effort has been devoted in this research area to improve the generation quality or enhance the model generalization. However, there are few works able to address both issues simultaneously, which is essential for practical applications. To this end, in this paper, we turn attention to the emerging powerful Latent Diffusion Models, and model the Talking head generation as an audio-driven temporally coherent denoising process (DiffTalk). More specifically, instead of employing audio signals as th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.03786","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-01-10T05:11:25Z","cross_cats_sorted":[],"title_canon_sha256":"5db788a31c5c0682de15a7d7e9b86e0b4456282cd407947386a0c947b6e78019","abstract_canon_sha256":"3ccbe6a6f0151b265c91fb826f0775995a4180eec3b757859b17344219d65c34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:02:48.929174Z","signature_b64":"vP5hEeriOCFdyF/asxfOIvM2HAY9qM8ElxpwmBFT3YQdLItgGvoWxSpCtAx6tDtpai87r2iQDi9o4fdB1BXYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af910198f8b497197ba2824ac06fd16d720093d4324318eb58264ceadb199bcb","last_reissued_at":"2026-07-05T06:02:48.928685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:02:48.928685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiffTalk: Crafting Diffusion Models for Generalized Audio-Driven Portraits Animation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jie Zhou, Jiwen Lu, Shuai Shen, Wanhua Li, Wenliang Zhao, Zheng Zhu, Zibin Meng","submitted_at":"2023-01-10T05:11:25Z","abstract_excerpt":"Talking head synthesis is a promising approach for the video production industry. Recently, a lot of effort has been devoted in this research area to improve the generation quality or enhance the model generalization. However, there are few works able to address both issues simultaneously, which is essential for practical applications. To this end, in this paper, we turn attention to the emerging powerful Latent Diffusion Models, and model the Talking head generation as an audio-driven temporally coherent denoising process (DiffTalk). More specifically, instead of employing audio signals as th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.03786","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.03786/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.03786","created_at":"2026-07-05T06:02:48.928750+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.03786v2","created_at":"2026-07-05T06:02:48.928750+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.03786","created_at":"2026-07-05T06:02:48.928750+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6IQDGHYWSLR","created_at":"2026-07-05T06:02:48.928750+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6IQDGHYWSLRS65C","created_at":"2026-07-05T06:02:48.928750+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6IQDGHY","created_at":"2026-07-05T06:02:48.928750+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01031","citing_title":"Temporally-Aligned Evaluation for Audio-Driven Talking Head Generation","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV","json":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV.json","graph_json":"https://pith.science/api/pith-number/V6IQDGHYWSLRS65CQJFMA36RNV/graph.json","events_json":"https://pith.science/api/pith-number/V6IQDGHYWSLRS65CQJFMA36RNV/events.json","paper":"https://pith.science/paper/V6IQDGHY"},"agent_actions":{"view_html":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV","download_json":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV.json","view_paper":"https://pith.science/paper/V6IQDGHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.03786&json=true","fetch_graph":"https://pith.science/api/pith-number/V6IQDGHYWSLRS65CQJFMA36RNV/graph.json","fetch_events":"https://pith.science/api/pith-number/V6IQDGHYWSLRS65CQJFMA36RNV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV/action/storage_attestation","attest_author":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV/action/author_attestation","sign_citation":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV/action/citation_signature","submit_replication":"https://pith.science/pith/V6IQDGHYWSLRS65CQJFMA36RNV/action/replication_record"}},"created_at":"2026-07-05T06:02:48.928750+00:00","updated_at":"2026-07-05T06:02:48.928750+00:00"}