{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4UONZEG6VDY6OO3DCJ5IMSNGPB","short_pith_number":"pith:4UONZEG6","schema_version":"1.0","canonical_sha256":"e51cdc90dea8f1e73b63127a8649a678469601edfcc227c80f2f4214fe7174b1","source":{"kind":"arxiv","id":"2502.04847","version":5},"attestation_state":"computed","paper":{"title":"HumanDiT: Pose-Guided Diffusion Transformer for Long-form Human Motion Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyue Peng, Chen Zhang, Jianke Zhu, Pan Xie, Qijun Gan, Xiang Yin, Yi Ren, Zehuan Yuan, Zhenhui Ye","submitted_at":"2025-02-07T11:36:36Z","abstract_excerpt":"Human motion video generation has advanced significantly, while existing methods still struggle with accurately rendering detailed body parts like hands and faces, especially in long sequences and intricate motions. Current approaches also rely on fixed resolution and struggle to maintain visual consistency. To address these limitations, we propose HumanDiT, a pose-guided Diffusion Transformer (DiT)-based framework trained on a large and wild dataset containing 14,000 hours of high-quality video to produce high-fidelity videos with fine-grained body rendering. Specifically, (i) HumanDiT, built"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.04847","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-07T11:36:36Z","cross_cats_sorted":[],"title_canon_sha256":"a872184bf32d08e105c98a720ed93036d23ce4c7e818a08a0003692f13618a56","abstract_canon_sha256":"68eca88dcc3bad3ff0e1176e2bf159aa8f09b894259d23a2f7a01d81c948368b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:00.524684Z","signature_b64":"Z6LUCJ3wBXJ8YXiqBhTJeVyPdo2QWKwex5jGBLDVNh49YZsItm/Qfba2N7BJRzTCVnYd1EIlyYGkaTwtm6gzBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e51cdc90dea8f1e73b63127a8649a678469601edfcc227c80f2f4214fe7174b1","last_reissued_at":"2026-07-05T11:48:00.524100Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:00.524100Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HumanDiT: Pose-Guided Diffusion Transformer for Long-form Human Motion Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyue Peng, Chen Zhang, Jianke Zhu, Pan Xie, Qijun Gan, Xiang Yin, Yi Ren, Zehuan Yuan, Zhenhui Ye","submitted_at":"2025-02-07T11:36:36Z","abstract_excerpt":"Human motion video generation has advanced significantly, while existing methods still struggle with accurately rendering detailed body parts like hands and faces, especially in long sequences and intricate motions. Current approaches also rely on fixed resolution and struggle to maintain visual consistency. To address these limitations, we propose HumanDiT, a pose-guided Diffusion Transformer (DiT)-based framework trained on a large and wild dataset containing 14,000 hours of high-quality video to produce high-fidelity videos with fine-grained body rendering. Specifically, (i) HumanDiT, built"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.04847","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.04847/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.04847","created_at":"2026-07-05T11:48:00.524218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.04847v5","created_at":"2026-07-05T11:48:00.524218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.04847","created_at":"2026-07-05T11:48:00.524218+00:00"},{"alias_kind":"pith_short_12","alias_value":"4UONZEG6VDY6","created_at":"2026-07-05T11:48:00.524218+00:00"},{"alias_kind":"pith_short_16","alias_value":"4UONZEG6VDY6OO3D","created_at":"2026-07-05T11:48:00.524218+00:00"},{"alias_kind":"pith_short_8","alias_value":"4UONZEG6","created_at":"2026-07-05T11:48:00.524218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06903","citing_title":"Beyond Skeletons: Learning Animation Directly from Driving Videos with Same2X Training Strategy","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02000","citing_title":"Towards 3D-Aware Video Diffusion Models: Render-Free Human Motion Control with Mesh Tokenization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15042","citing_title":"EverAnimate: Minute-Scale Human Animation via Latent Flow Restoration","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29020","citing_title":"Semantic-Aware, Physics-Informed, Geometry-Grounded Weather Video Synthesis","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22940","citing_title":"One-to-All Animation: Alignment-Free Character Animation and Image Pose Transfer","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17248","citing_title":"Image-to-Video Diffusion: From Foundations to Open Frontiers","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10632","citing_title":"CoMoVi: Co-Generation of 3D Human Motions and Realistic Videos","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11804","citing_title":"OmniShow: Unifying Multimodal Conditions for Human-Object Interaction Video Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21776","citing_title":"Reshoot-Anything: A Self-Supervised Model for In-the-Wild Video Reshooting","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB","json":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB.json","graph_json":"https://pith.science/api/pith-number/4UONZEG6VDY6OO3DCJ5IMSNGPB/graph.json","events_json":"https://pith.science/api/pith-number/4UONZEG6VDY6OO3DCJ5IMSNGPB/events.json","paper":"https://pith.science/paper/4UONZEG6"},"agent_actions":{"view_html":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB","download_json":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB.json","view_paper":"https://pith.science/paper/4UONZEG6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.04847&json=true","fetch_graph":"https://pith.science/api/pith-number/4UONZEG6VDY6OO3DCJ5IMSNGPB/graph.json","fetch_events":"https://pith.science/api/pith-number/4UONZEG6VDY6OO3DCJ5IMSNGPB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB/action/storage_attestation","attest_author":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB/action/author_attestation","sign_citation":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB/action/citation_signature","submit_replication":"https://pith.science/pith/4UONZEG6VDY6OO3DCJ5IMSNGPB/action/replication_record"}},"created_at":"2026-07-05T11:48:00.524218+00:00","updated_at":"2026-07-05T11:48:00.524218+00:00"}