{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:P7KHYNGCLRKBS4B6EWY4IDB45Y","short_pith_number":"pith:P7KHYNGC","schema_version":"1.0","canonical_sha256":"7fd47c34c25c5419703e25b1c40c3cee2e8cf0dee33a2d5ee0ddaa093be07b12","source":{"kind":"arxiv","id":"2503.09942","version":1},"attestation_state":"computed","paper":{"title":"Cosh-DiT: Co-Speech Gesture Video Synthesis via Hybrid Audio-Visual Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Borong Liang, Hang Zhou, Haocheng Feng, Jiazhi Guan, Jingdong Wang, Kaisiyuan Wang, Koike Hideki, Quanwei Yang, Yasheng Sun, Yingying Li, Zhiliang Xu, Ziwei Liu","submitted_at":"2025-03-13T01:36:05Z","abstract_excerpt":"Co-speech gesture video synthesis is a challenging task that requires both probabilistic modeling of human gestures and the synthesis of realistic images that align with the rhythmic nuances of speech. To address these challenges, we propose Cosh-DiT, a Co-speech gesture video system with hybrid Diffusion Transformers that perform audio-to-motion and motion-to-video synthesis using discrete and continuous diffusion modeling, respectively. First, we introduce an audio Diffusion Transformer (Cosh-DiT-A) to synthesize expressive gesture dynamics synchronized with speech rhythms. To capture upper "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.09942","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T01:36:05Z","cross_cats_sorted":[],"title_canon_sha256":"1d72153fc9415ba09b20e94b5aa556a6b5946826c397a9029ef38eaef5b5b73f","abstract_canon_sha256":"290b3388bbd50661fce6e7cf6543342d961defb32dc1ec332d28ce0f38d726e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:35.281762Z","signature_b64":"gP159g6TaK0cdf35gHp4tD7ODjylb5QL+AB+us5NrNJX7Te+ZAuMPkjGY5qHo2dhX2GcsGwAxZJC6J6vYyvBAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fd47c34c25c5419703e25b1c40c3cee2e8cf0dee33a2d5ee0ddaa093be07b12","last_reissued_at":"2026-07-05T10:30:35.280811Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:35.280811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cosh-DiT: Co-Speech Gesture Video Synthesis via Hybrid Audio-Visual Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Borong Liang, Hang Zhou, Haocheng Feng, Jiazhi Guan, Jingdong Wang, Kaisiyuan Wang, Koike Hideki, Quanwei Yang, Yasheng Sun, Yingying Li, Zhiliang Xu, Ziwei Liu","submitted_at":"2025-03-13T01:36:05Z","abstract_excerpt":"Co-speech gesture video synthesis is a challenging task that requires both probabilistic modeling of human gestures and the synthesis of realistic images that align with the rhythmic nuances of speech. To address these challenges, we propose Cosh-DiT, a Co-speech gesture video system with hybrid Diffusion Transformers that perform audio-to-motion and motion-to-video synthesis using discrete and continuous diffusion modeling, respectively. First, we introduce an audio Diffusion Transformer (Cosh-DiT-A) to synthesize expressive gesture dynamics synchronized with speech rhythms. To capture upper "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.09942","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.09942/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.09942","created_at":"2026-07-05T10:30:35.280917+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.09942v1","created_at":"2026-07-05T10:30:35.280917+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.09942","created_at":"2026-07-05T10:30:35.280917+00:00"},{"alias_kind":"pith_short_12","alias_value":"P7KHYNGCLRKB","created_at":"2026-07-05T10:30:35.280917+00:00"},{"alias_kind":"pith_short_16","alias_value":"P7KHYNGCLRKBS4B6","created_at":"2026-07-05T10:30:35.280917+00:00"},{"alias_kind":"pith_short_8","alias_value":"P7KHYNGC","created_at":"2026-07-05T10:30:35.280917+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y","json":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y.json","graph_json":"https://pith.science/api/pith-number/P7KHYNGCLRKBS4B6EWY4IDB45Y/graph.json","events_json":"https://pith.science/api/pith-number/P7KHYNGCLRKBS4B6EWY4IDB45Y/events.json","paper":"https://pith.science/paper/P7KHYNGC"},"agent_actions":{"view_html":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y","download_json":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y.json","view_paper":"https://pith.science/paper/P7KHYNGC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.09942&json=true","fetch_graph":"https://pith.science/api/pith-number/P7KHYNGCLRKBS4B6EWY4IDB45Y/graph.json","fetch_events":"https://pith.science/api/pith-number/P7KHYNGCLRKBS4B6EWY4IDB45Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y/action/storage_attestation","attest_author":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y/action/author_attestation","sign_citation":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y/action/citation_signature","submit_replication":"https://pith.science/pith/P7KHYNGCLRKBS4B6EWY4IDB45Y/action/replication_record"}},"created_at":"2026-07-05T10:30:35.280917+00:00","updated_at":"2026-07-05T10:30:35.280917+00:00"}