{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZNOAKPXSPFVAWLVI3NS2Q66A7O","short_pith_number":"pith:ZNOAKPXS","schema_version":"1.0","canonical_sha256":"cb5c053ef2796a0b2ea8db65a87bc0fb9a33f0e8cfff87beb36caef539c2ca62","source":{"kind":"arxiv","id":"2307.09368","version":3},"attestation_state":"computed","paper":{"title":"Audio-driven Talking Face Generation with Stabilized Synchronization Loss","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexander Waibel, Dogucan Yaman, Fevziye Irem Eyiokur, Hazim Kemal Ekenel, Leonard B\\\"armann","submitted_at":"2023-07-18T15:50:04Z","abstract_excerpt":"Talking face generation aims to create realistic videos with accurate lip synchronization and high visual quality, using given audio and reference video while preserving identity and visual characteristics. In this paper, we start by identifying several issues with existing synchronization learning methods. These involve unstable training, lip synchronization, and visual quality issues caused by lip-sync loss, SyncNet, and lip leaking from the identity reference. To address these issues, we first tackle the lip leaking problem by introducing a silent-lip generator, which changes the lips of th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.09368","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-07-18T15:50:04Z","cross_cats_sorted":[],"title_canon_sha256":"f4a8ce77ff4339ae379a98194587b50ccd10f874e3991f52bd2b39ac8161530f","abstract_canon_sha256":"2508f982c302581515f7991d0cfcb0073661e3169ba0022564418f6718c19c09"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:21.375994Z","signature_b64":"Xbh8mFGtD2QSUkH0jrI8U720kPbiIXAMIJyoH0wx5leq3XI4TKhdeDuKlYssqIUctQCUpj/c16FfoljqtE1ZBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb5c053ef2796a0b2ea8db65a87bc0fb9a33f0e8cfff87beb36caef539c2ca62","last_reissued_at":"2026-07-05T08:45:21.375488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:21.375488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-driven Talking Face Generation with Stabilized Synchronization Loss","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexander Waibel, Dogucan Yaman, Fevziye Irem Eyiokur, Hazim Kemal Ekenel, Leonard B\\\"armann","submitted_at":"2023-07-18T15:50:04Z","abstract_excerpt":"Talking face generation aims to create realistic videos with accurate lip synchronization and high visual quality, using given audio and reference video while preserving identity and visual characteristics. In this paper, we start by identifying several issues with existing synchronization learning methods. These involve unstable training, lip synchronization, and visual quality issues caused by lip-sync loss, SyncNet, and lip leaking from the identity reference. To address these issues, we first tackle the lip leaking problem by introducing a silent-lip generator, which changes the lips of th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.09368","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.09368/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.09368","created_at":"2026-07-05T08:45:21.375550+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.09368v3","created_at":"2026-07-05T08:45:21.375550+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.09368","created_at":"2026-07-05T08:45:21.375550+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZNOAKPXSPFVA","created_at":"2026-07-05T08:45:21.375550+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZNOAKPXSPFVAWLVI","created_at":"2026-07-05T08:45:21.375550+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZNOAKPXS","created_at":"2026-07-05T08:45:21.375550+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.20953","citing_title":"Mask-Free Audio-driven Talking Face Generation for Enhanced Visual Quality and Identity Preservation","ref_index":94,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O","json":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O.json","graph_json":"https://pith.science/api/pith-number/ZNOAKPXSPFVAWLVI3NS2Q66A7O/graph.json","events_json":"https://pith.science/api/pith-number/ZNOAKPXSPFVAWLVI3NS2Q66A7O/events.json","paper":"https://pith.science/paper/ZNOAKPXS"},"agent_actions":{"view_html":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O","download_json":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O.json","view_paper":"https://pith.science/paper/ZNOAKPXS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.09368&json=true","fetch_graph":"https://pith.science/api/pith-number/ZNOAKPXSPFVAWLVI3NS2Q66A7O/graph.json","fetch_events":"https://pith.science/api/pith-number/ZNOAKPXSPFVAWLVI3NS2Q66A7O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O/action/storage_attestation","attest_author":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O/action/author_attestation","sign_citation":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O/action/citation_signature","submit_replication":"https://pith.science/pith/ZNOAKPXSPFVAWLVI3NS2Q66A7O/action/replication_record"}},"created_at":"2026-07-05T08:45:21.375550+00:00","updated_at":"2026-07-05T08:45:21.375550+00:00"}