{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M2UNT6IRU567H3GFOKIN3C4OJH","short_pith_number":"pith:M2UNT6IR","schema_version":"1.0","canonical_sha256":"66a8d9f911a77df3ecc57290dd8b8e49c98ba49e7d508e375c802a0e9c58c16a","source":{"kind":"arxiv","id":"2407.05577","version":1},"attestation_state":"computed","paper":{"title":"Audio-driven High-resolution Seamless Talking Head Video Editing via StyleGAN","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongdong Lv, Jiacheng Su, Junfeng Yao, Kunhong Liu, Liyan Chen, Qingsong Liu","submitted_at":"2024-07-08T03:17:10Z","abstract_excerpt":"The existing methods for audio-driven talking head video editing have the limitations of poor visual effects. This paper tries to tackle this problem through editing talking face images seamless with different emotions based on two modules: (1) an audio-to-landmark module, consisting of the CrossReconstructed Emotion Disentanglement and an alignment network module. It bridges the gap between speech and facial motions by predicting corresponding emotional landmarks from speech; (2) a landmark-based editing module edits face videos via StyleGAN. It aims to generate the seamless edited video cons"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05577","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-08T03:17:10Z","cross_cats_sorted":[],"title_canon_sha256":"c3e5b119f259a0e6257ec851febc70334d0cb0b1d10ed5af5f775637d15c1f21","abstract_canon_sha256":"a8b9cca2b585d336d8204220f03b61b56105ffc0dd8c926848cbc674ae1c81b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:15.580058Z","signature_b64":"tAt7aZD9G3WPIhR7xA+T4OQHqFeEqfD8GIiRJ2z200B+h35o1TS8y4GSswab0ImEs4k20itw80w5vLvyh4+BCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66a8d9f911a77df3ecc57290dd8b8e49c98ba49e7d508e375c802a0e9c58c16a","last_reissued_at":"2026-07-05T08:41:15.579580Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:15.579580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-driven High-resolution Seamless Talking Head Video Editing via StyleGAN","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongdong Lv, Jiacheng Su, Junfeng Yao, Kunhong Liu, Liyan Chen, Qingsong Liu","submitted_at":"2024-07-08T03:17:10Z","abstract_excerpt":"The existing methods for audio-driven talking head video editing have the limitations of poor visual effects. This paper tries to tackle this problem through editing talking face images seamless with different emotions based on two modules: (1) an audio-to-landmark module, consisting of the CrossReconstructed Emotion Disentanglement and an alignment network module. It bridges the gap between speech and facial motions by predicting corresponding emotional landmarks from speech; (2) a landmark-based editing module edits face videos via StyleGAN. It aims to generate the seamless edited video cons"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05577","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05577/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05577","created_at":"2026-07-05T08:41:15.579645+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05577v1","created_at":"2026-07-05T08:41:15.579645+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05577","created_at":"2026-07-05T08:41:15.579645+00:00"},{"alias_kind":"pith_short_12","alias_value":"M2UNT6IRU567","created_at":"2026-07-05T08:41:15.579645+00:00"},{"alias_kind":"pith_short_16","alias_value":"M2UNT6IRU567H3GF","created_at":"2026-07-05T08:41:15.579645+00:00"},{"alias_kind":"pith_short_8","alias_value":"M2UNT6IR","created_at":"2026-07-05T08:41:15.579645+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00830","citing_title":"SkyReels-Audio: Omni Audio-Conditioned Talking Portraits in Video Diffusion Transformers","ref_index":57,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH","json":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH.json","graph_json":"https://pith.science/api/pith-number/M2UNT6IRU567H3GFOKIN3C4OJH/graph.json","events_json":"https://pith.science/api/pith-number/M2UNT6IRU567H3GFOKIN3C4OJH/events.json","paper":"https://pith.science/paper/M2UNT6IR"},"agent_actions":{"view_html":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH","download_json":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH.json","view_paper":"https://pith.science/paper/M2UNT6IR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05577&json=true","fetch_graph":"https://pith.science/api/pith-number/M2UNT6IRU567H3GFOKIN3C4OJH/graph.json","fetch_events":"https://pith.science/api/pith-number/M2UNT6IRU567H3GFOKIN3C4OJH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH/action/storage_attestation","attest_author":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH/action/author_attestation","sign_citation":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH/action/citation_signature","submit_replication":"https://pith.science/pith/M2UNT6IRU567H3GFOKIN3C4OJH/action/replication_record"}},"created_at":"2026-07-05T08:41:15.579645+00:00","updated_at":"2026-07-05T08:41:15.579645+00:00"}