{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FCBD62FOXLYMX5AIO2YHRU3VZ4","short_pith_number":"pith:FCBD62FO","schema_version":"1.0","canonical_sha256":"28823f68aebaf0cbf40876b078d375cf0306a30a3a86fb87c5449281be3d2ab6","source":{"kind":"arxiv","id":"2409.02657","version":1},"attestation_state":"computed","paper":{"title":"PoseTalk: Text-and-Audio-based Pose Control and Motion Refinement for One-Shot Talking Head Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Han Xue, Jun Ling, Li Song, Rong Xie, Yiwen Wang","submitted_at":"2024-09-04T12:30:25Z","abstract_excerpt":"While previous audio-driven talking head generation (THG) methods generate head poses from driving audio, the generated poses or lips cannot match the audio well or are not editable. In this study, we propose \\textbf{PoseTalk}, a THG system that can freely generate lip-synchronized talking head videos with free head poses conditioned on text prompts and audio. The core insight of our method is using head pose to connect visual, linguistic, and audio signals. First, we propose to generate poses from both audio and text prompts, where the audio offers short-term variations and rhythm corresponde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.02657","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-04T12:30:25Z","cross_cats_sorted":["cs.AI","cs.MM"],"title_canon_sha256":"d39c6a0e8460fd5224bde352c2196abd124689664272d04a9f2743ebc90c5d0d","abstract_canon_sha256":"0e6e3e11a54f67d9ab6ee58513eb07d8d556366bc14a9f9997f78fb2335b42c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:03:08.345483Z","signature_b64":"jc4cJZMDhXLThuIwLsKpnPvcXyUfoM4qcrrj7RT34+jAbeDbUxnGSHMqMTMagjlEtAVeReqA0qZpFWc3TUFnAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28823f68aebaf0cbf40876b078d375cf0306a30a3a86fb87c5449281be3d2ab6","last_reissued_at":"2026-07-05T09:03:08.344998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:03:08.344998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PoseTalk: Text-and-Audio-based Pose Control and Motion Refinement for One-Shot Talking Head Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Han Xue, Jun Ling, Li Song, Rong Xie, Yiwen Wang","submitted_at":"2024-09-04T12:30:25Z","abstract_excerpt":"While previous audio-driven talking head generation (THG) methods generate head poses from driving audio, the generated poses or lips cannot match the audio well or are not editable. In this study, we propose \\textbf{PoseTalk}, a THG system that can freely generate lip-synchronized talking head videos with free head poses conditioned on text prompts and audio. The core insight of our method is using head pose to connect visual, linguistic, and audio signals. First, we propose to generate poses from both audio and text prompts, where the audio offers short-term variations and rhythm corresponde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.02657","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.02657/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.02657","created_at":"2026-07-05T09:03:08.345070+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.02657v1","created_at":"2026-07-05T09:03:08.345070+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.02657","created_at":"2026-07-05T09:03:08.345070+00:00"},{"alias_kind":"pith_short_12","alias_value":"FCBD62FOXLYM","created_at":"2026-07-05T09:03:08.345070+00:00"},{"alias_kind":"pith_short_16","alias_value":"FCBD62FOXLYMX5AI","created_at":"2026-07-05T09:03:08.345070+00:00"},{"alias_kind":"pith_short_8","alias_value":"FCBD62FO","created_at":"2026-07-05T09:03:08.345070+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20861","citing_title":"Exploring Timeline Control for Facial Motion Generation","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4","json":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4.json","graph_json":"https://pith.science/api/pith-number/FCBD62FOXLYMX5AIO2YHRU3VZ4/graph.json","events_json":"https://pith.science/api/pith-number/FCBD62FOXLYMX5AIO2YHRU3VZ4/events.json","paper":"https://pith.science/paper/FCBD62FO"},"agent_actions":{"view_html":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4","download_json":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4.json","view_paper":"https://pith.science/paper/FCBD62FO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.02657&json=true","fetch_graph":"https://pith.science/api/pith-number/FCBD62FOXLYMX5AIO2YHRU3VZ4/graph.json","fetch_events":"https://pith.science/api/pith-number/FCBD62FOXLYMX5AIO2YHRU3VZ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4/action/storage_attestation","attest_author":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4/action/author_attestation","sign_citation":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4/action/citation_signature","submit_replication":"https://pith.science/pith/FCBD62FOXLYMX5AIO2YHRU3VZ4/action/replication_record"}},"created_at":"2026-07-05T09:03:08.345070+00:00","updated_at":"2026-07-05T09:03:08.345070+00:00"}