{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XG3CWBB27VPFDSEGRLTZ6Z37CY","short_pith_number":"pith:XG3CWBB2","schema_version":"1.0","canonical_sha256":"b9b62b043afd5e51c8868ae79f677f163bea768e0ff8eb057919452206d21961","source":{"kind":"arxiv","id":"2002.10137","version":2},"attestation_state":"computed","paper":{"title":"Audio-driven Talking Face Video Generation with Learning-based Personalized Head Pose","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Hujun Bao, Juyong Zhang, Ran Yi, Yong-jin Liu, Zipeng Ye","submitted_at":"2020-02-24T10:02:10Z","abstract_excerpt":"Real-world talking faces often accompany with natural head movement. However, most existing talking face video generation methods only consider facial animation with fixed head pose. In this paper, we address this problem by proposing a deep neural network model that takes an audio signal A of a source person and a very short video V of a target person as input, and outputs a synthesized high-quality talking face video with personalized head pose (making use of the visual information in V), expression and lip synchronization (by considering both A and V). The most challenging issue in our work"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.10137","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-02-24T10:02:10Z","cross_cats_sorted":["cs.GR"],"title_canon_sha256":"6d3dce42ce3165ae1f2ed521da71cc918cb47c0171139caedf99d12d95d0a64d","abstract_canon_sha256":"2741022a2ba291df8238199b514cf728c42e12638e147af3fecfbc3fa5967c35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:45:54.339915Z","signature_b64":"3WRnc5XWvZTGQWFUwTrxMuvdpAXT9qGk8q309tyRc/EQrCHxqEBRGT7EyFTPE1Vli1PYXEetss12gSyJnuZADw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9b62b043afd5e51c8868ae79f677f163bea768e0ff8eb057919452206d21961","last_reissued_at":"2026-07-05T00:45:54.339495Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:45:54.339495Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-driven Talking Face Video Generation with Learning-based Personalized Head Pose","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Hujun Bao, Juyong Zhang, Ran Yi, Yong-jin Liu, Zipeng Ye","submitted_at":"2020-02-24T10:02:10Z","abstract_excerpt":"Real-world talking faces often accompany with natural head movement. However, most existing talking face video generation methods only consider facial animation with fixed head pose. In this paper, we address this problem by proposing a deep neural network model that takes an audio signal A of a source person and a very short video V of a target person as input, and outputs a synthesized high-quality talking face video with personalized head pose (making use of the visual information in V), expression and lip synchronization (by considering both A and V). The most challenging issue in our work"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.10137","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.10137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.10137","created_at":"2026-07-05T00:45:54.339555+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.10137v2","created_at":"2026-07-05T00:45:54.339555+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.10137","created_at":"2026-07-05T00:45:54.339555+00:00"},{"alias_kind":"pith_short_12","alias_value":"XG3CWBB27VPF","created_at":"2026-07-05T00:45:54.339555+00:00"},{"alias_kind":"pith_short_16","alias_value":"XG3CWBB27VPFDSEG","created_at":"2026-07-05T00:45:54.339555+00:00"},{"alias_kind":"pith_short_8","alias_value":"XG3CWBB2","created_at":"2026-07-05T00:45:54.339555+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25488","citing_title":"Test-Time Self-Adaptive Conditioning for Stable Audio-Driven Talking-Head Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02941","citing_title":"MMTalker: Multiresolution 3D Talking Head Synthesis with Multimodal Feature Fusion","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY","json":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY.json","graph_json":"https://pith.science/api/pith-number/XG3CWBB27VPFDSEGRLTZ6Z37CY/graph.json","events_json":"https://pith.science/api/pith-number/XG3CWBB27VPFDSEGRLTZ6Z37CY/events.json","paper":"https://pith.science/paper/XG3CWBB2"},"agent_actions":{"view_html":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY","download_json":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY.json","view_paper":"https://pith.science/paper/XG3CWBB2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.10137&json=true","fetch_graph":"https://pith.science/api/pith-number/XG3CWBB27VPFDSEGRLTZ6Z37CY/graph.json","fetch_events":"https://pith.science/api/pith-number/XG3CWBB27VPFDSEGRLTZ6Z37CY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY/action/storage_attestation","attest_author":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY/action/author_attestation","sign_citation":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY/action/citation_signature","submit_replication":"https://pith.science/pith/XG3CWBB27VPFDSEGRLTZ6Z37CY/action/replication_record"}},"created_at":"2026-07-05T00:45:54.339555+00:00","updated_at":"2026-07-05T00:45:54.339555+00:00"}