{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:35LX3C7ADHBCVKG2J7Z3E7TTOI","short_pith_number":"pith:35LX3C7A","schema_version":"1.0","canonical_sha256":"df577d8be019c22aa8da4ff3b27e73723bdc7cff39a6ec1dea0e4ca6b864c22b","source":{"kind":"arxiv","id":"2407.08136","version":2},"attestation_state":"computed","paper":{"title":"EchoMimic: Lifelike Audio-Driven Portrait Animations through Editable Landmark Conditions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenguang Ma, Jiajiong Cao, Yuming Li, Zhiquan Chen, Zhiyuan Chen","submitted_at":"2024-07-11T02:26:51Z","abstract_excerpt":"The area of portrait image animation, propelled by audio input, has witnessed notable progress in the generation of lifelike and dynamic portraits. Conventional methods are limited to utilizing either audios or facial key points to drive images into videos, while they can yield satisfactory results, certain issues exist. For instance, methods driven solely by audios can be unstable at times due to the relatively weaker audio signal, while methods driven exclusively by facial key points, although more stable in driving, can result in unnatural outcomes due to the excessive control of key point "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08136","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-11T02:26:51Z","cross_cats_sorted":[],"title_canon_sha256":"e31b35386b9d7e063b186924d9e0ec1361f553590c72634ec2ed346f69dea882","abstract_canon_sha256":"622c96d24a0d93ad3a7338bcdcf6173ae7de16ac1206e485ec7c00f70e105417"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:55.366920Z","signature_b64":"n3TgU7uCtbnyP/dvrnP0QV5K7tjthJZuilxtbXu1IXlKzPmJpmGjryf9wTQ5bH6Q3xuiOocAOWaExAxXTXvNAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df577d8be019c22aa8da4ff3b27e73723bdc7cff39a6ec1dea0e4ca6b864c22b","last_reissued_at":"2026-07-05T08:42:55.366529Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:55.366529Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EchoMimic: Lifelike Audio-Driven Portrait Animations through Editable Landmark Conditions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenguang Ma, Jiajiong Cao, Yuming Li, Zhiquan Chen, Zhiyuan Chen","submitted_at":"2024-07-11T02:26:51Z","abstract_excerpt":"The area of portrait image animation, propelled by audio input, has witnessed notable progress in the generation of lifelike and dynamic portraits. Conventional methods are limited to utilizing either audios or facial key points to drive images into videos, while they can yield satisfactory results, certain issues exist. For instance, methods driven solely by audios can be unstable at times due to the relatively weaker audio signal, while methods driven exclusively by facial key points, although more stable in driving, can result in unnatural outcomes due to the excessive control of key point "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08136","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08136/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08136","created_at":"2026-07-05T08:42:55.366585+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08136v2","created_at":"2026-07-05T08:42:55.366585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08136","created_at":"2026-07-05T08:42:55.366585+00:00"},{"alias_kind":"pith_short_12","alias_value":"35LX3C7ADHBC","created_at":"2026-07-05T08:42:55.366585+00:00"},{"alias_kind":"pith_short_16","alias_value":"35LX3C7ADHBCVKG2","created_at":"2026-07-05T08:42:55.366585+00:00"},{"alias_kind":"pith_short_8","alias_value":"35LX3C7A","created_at":"2026-07-05T08:42:55.366585+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.16748","citing_title":"Multimodal Diffusion Transformer with Memory Bank for Scalable Long-Duration Talking Video Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03603","citing_title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16918","citing_title":"HighSync: High-Quality Lip Synchronization via Latent Diffusion Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13669","citing_title":"EchoTorrent: Towards Swift, Sustained, and Streaming Multi-Modal Video Generation","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23632","citing_title":"Hallo-Live: Real-Time Streaming Joint Audio-Video Avatar Generation with Asynchronous Dual-Stream and Human-Centric Preference Distillation","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI","json":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI.json","graph_json":"https://pith.science/api/pith-number/35LX3C7ADHBCVKG2J7Z3E7TTOI/graph.json","events_json":"https://pith.science/api/pith-number/35LX3C7ADHBCVKG2J7Z3E7TTOI/events.json","paper":"https://pith.science/paper/35LX3C7A"},"agent_actions":{"view_html":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI","download_json":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI.json","view_paper":"https://pith.science/paper/35LX3C7A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08136&json=true","fetch_graph":"https://pith.science/api/pith-number/35LX3C7ADHBCVKG2J7Z3E7TTOI/graph.json","fetch_events":"https://pith.science/api/pith-number/35LX3C7ADHBCVKG2J7Z3E7TTOI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI/action/storage_attestation","attest_author":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI/action/author_attestation","sign_citation":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI/action/citation_signature","submit_replication":"https://pith.science/pith/35LX3C7ADHBCVKG2J7Z3E7TTOI/action/replication_record"}},"created_at":"2026-07-05T08:42:55.366585+00:00","updated_at":"2026-07-05T08:42:55.366585+00:00"}