{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CVPT3DJ3J3J6VC7GGGMTM4SIQP","short_pith_number":"pith:CVPT3DJ3","schema_version":"1.0","canonical_sha256":"155f3d8d3b4ed3ea8be6319936724883e76ff25440a741aeccf306928129abe7","source":{"kind":"arxiv","id":"2308.07749","version":1},"attestation_state":"computed","paper":{"title":"Dancing Avatar: Pose and Text-Guided Human Motion Videos Synthesis with Image Diffusion Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bosheng Qin, Qifan Yu, Siliang Tang, Wentao Ye, Yueting Zhuang","submitted_at":"2023-08-15T13:00:42Z","abstract_excerpt":"The rising demand for creating lifelike avatars in the digital realm has led to an increased need for generating high-quality human videos guided by textual descriptions and poses. We propose Dancing Avatar, designed to fabricate human motion videos driven by poses and textual cues. Our approach employs a pretrained T2I diffusion model to generate each video frame in an autoregressive fashion. The crux of innovation lies in our adept utilization of the T2I diffusion model for producing video frames successively while preserving contextual relevance. We surmount the hurdles posed by maintaining"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.07749","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-15T13:00:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9dbcf12d18f8e56dfb71a008f2531c762ecd3dd6a6196bed2298487019de0aa9","abstract_canon_sha256":"aa45c94955d018c7e3b0167e84962c3eddf24fcce2230b1ce3fc3cec16325f34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:41:30.555869Z","signature_b64":"nh6UO4iLk0HxF47hfnOXdDbUBtY/9r9vokXZuKK/pcQo88z8Bq6EMGvIVBf1LsQDe54NdjTECvzsFQMDBiIGAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"155f3d8d3b4ed3ea8be6319936724883e76ff25440a741aeccf306928129abe7","last_reissued_at":"2026-07-05T06:41:30.555330Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:41:30.555330Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dancing Avatar: Pose and Text-Guided Human Motion Videos Synthesis with Image Diffusion Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bosheng Qin, Qifan Yu, Siliang Tang, Wentao Ye, Yueting Zhuang","submitted_at":"2023-08-15T13:00:42Z","abstract_excerpt":"The rising demand for creating lifelike avatars in the digital realm has led to an increased need for generating high-quality human videos guided by textual descriptions and poses. We propose Dancing Avatar, designed to fabricate human motion videos driven by poses and textual cues. Our approach employs a pretrained T2I diffusion model to generate each video frame in an autoregressive fashion. The crux of innovation lies in our adept utilization of the T2I diffusion model for producing video frames successively while preserving contextual relevance. We surmount the hurdles posed by maintaining"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.07749","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.07749/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.07749","created_at":"2026-07-05T06:41:30.555391+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.07749v1","created_at":"2026-07-05T06:41:30.555391+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.07749","created_at":"2026-07-05T06:41:30.555391+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVPT3DJ3J3J6","created_at":"2026-07-05T06:41:30.555391+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVPT3DJ3J3J6VC7G","created_at":"2026-07-05T06:41:30.555391+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVPT3DJ3","created_at":"2026-07-05T06:41:30.555391+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.21776","citing_title":"Reshoot-Anything: A Self-Supervised Model for In-the-Wild Video Reshooting","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP","json":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP.json","graph_json":"https://pith.science/api/pith-number/CVPT3DJ3J3J6VC7GGGMTM4SIQP/graph.json","events_json":"https://pith.science/api/pith-number/CVPT3DJ3J3J6VC7GGGMTM4SIQP/events.json","paper":"https://pith.science/paper/CVPT3DJ3"},"agent_actions":{"view_html":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP","download_json":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP.json","view_paper":"https://pith.science/paper/CVPT3DJ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.07749&json=true","fetch_graph":"https://pith.science/api/pith-number/CVPT3DJ3J3J6VC7GGGMTM4SIQP/graph.json","fetch_events":"https://pith.science/api/pith-number/CVPT3DJ3J3J6VC7GGGMTM4SIQP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP/action/storage_attestation","attest_author":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP/action/author_attestation","sign_citation":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP/action/citation_signature","submit_replication":"https://pith.science/pith/CVPT3DJ3J3J6VC7GGGMTM4SIQP/action/replication_record"}},"created_at":"2026-07-05T06:41:30.555391+00:00","updated_at":"2026-07-05T06:41:30.555391+00:00"}