{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ACKR4SU34PYA43YCRFKVLXQXXG","short_pith_number":"pith:ACKR4SU3","schema_version":"1.0","canonical_sha256":"00951e4a9be3f00e6f02895555de17b981021358a8113a71fd6d006ddcc963ac","source":{"kind":"arxiv","id":"2502.01061","version":3},"attestation_state":"computed","paper":{"title":"OmniHuman-1: Rethinking the Scaling-Up of One-Stage Conditioned Human Animation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chao Liang, Gaojie Lin, Jianwen Jiang, Jiaqi Yang, Zerong Zheng","submitted_at":"2025-02-03T05:17:32Z","abstract_excerpt":"End-to-end human animation, such as audio-driven talking human generation, has undergone notable advancements in the recent few years. However, existing methods still struggle to scale up as large general video generation models, limiting their potential in real applications. In this paper, we propose OmniHuman, a Diffusion Transformer-based framework that scales up data by mixing motion-related conditions into the training phase. To this end, we introduce two training principles for these mixed conditions, along with the corresponding model architecture and inference strategy. These designs e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01061","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-03T05:17:32Z","cross_cats_sorted":[],"title_canon_sha256":"d778d9691bb8c0878ac7cff466c8ac72ec3c0bbef37accb4551e874bb856dcf8","abstract_canon_sha256":"290aa5eacbb6071ef0c402654526d8d3e0eb8eb0d779b3b1e6467d93a062f7d7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:06.427024Z","signature_b64":"LdMvRB/3O60SoIpSgsc+hNr8pWzTYrpUHdsWQoPEOQewREesvy6BqORThinTK3mM839CsDy/b50s9WJB1dcTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00951e4a9be3f00e6f02895555de17b981021358a8113a71fd6d006ddcc963ac","last_reissued_at":"2026-07-05T11:29:06.426387Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:06.426387Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OmniHuman-1: Rethinking the Scaling-Up of One-Stage Conditioned Human Animation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chao Liang, Gaojie Lin, Jianwen Jiang, Jiaqi Yang, Zerong Zheng","submitted_at":"2025-02-03T05:17:32Z","abstract_excerpt":"End-to-end human animation, such as audio-driven talking human generation, has undergone notable advancements in the recent few years. However, existing methods still struggle to scale up as large general video generation models, limiting their potential in real applications. In this paper, we propose OmniHuman, a Diffusion Transformer-based framework that scales up data by mixing motion-related conditions into the training phase. To this end, we introduce two training principles for these mixed conditions, along with the corresponding model architecture and inference strategy. These designs e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01061","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01061/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01061","created_at":"2026-07-05T11:29:06.426467+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01061v3","created_at":"2026-07-05T11:29:06.426467+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01061","created_at":"2026-07-05T11:29:06.426467+00:00"},{"alias_kind":"pith_short_12","alias_value":"ACKR4SU34PYA","created_at":"2026-07-05T11:29:06.426467+00:00"},{"alias_kind":"pith_short_16","alias_value":"ACKR4SU34PYA43YC","created_at":"2026-07-05T11:29:06.426467+00:00"},{"alias_kind":"pith_short_8","alias_value":"ACKR4SU3","created_at":"2026-07-05T11:29:06.426467+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25041","citing_title":"Wan-Streamer v0.1: End-to-end Real-time Interactive Foundation Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25041","citing_title":"Wan-Streamer v0.1: End-to-end Real-time Interactive Foundation Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01677","citing_title":"ICDepth: Taming Video Diffusion Models for Video Depth Estimation via In-Context Conditioning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10839","citing_title":"HarmoView: Harmonizing Multi-View Constraints for Identity-Consistent Video Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30514","citing_title":"3D Scene-Adaptive Trajectory-Controllable Human Image Animation with Camera Movement","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02000","citing_title":"Towards 3D-Aware Video Diffusion Models: Render-Free Human Motion Control with Mesh Tokenization","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01620","citing_title":"Real-Time Generation of Streamable Talking Portrait Video with Reference-Guided Deep Compression VAEs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15042","citing_title":"EverAnimate: Minute-Scale Human Animation via Latent Flow Restoration","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25041","citing_title":"Wan-Streamer v0.1: End-to-end Real-time Interactive Foundation Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30514","citing_title":"3D Scene-Adaptive Trajectory-Controllable Human Image Animation with Camera Movement","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14505","citing_title":"MusicInfuser: Making Video Diffusion Listen and Dance","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23552","citing_title":"JAM-Flow: Joint Audio-Motion Synthesis with Flow Matching","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12052","citing_title":"FluentAvatar: Flicker-Free Talking-Head Animation via Phoneme-Guided Autoregressive Modeling","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12834","citing_title":"SAGA: Source Attribution of Generative AI Videos","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13669","citing_title":"EchoTorrent: Towards Swift, Sustained, and Streaming Multi-Modal Video Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01809","citing_title":"TMD-Bench: A Multi-Level Evaluation Paradigm for Music-Dance Co-Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07823","citing_title":"LPM 1.0: Video-based Character Performance Model","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG","json":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG.json","graph_json":"https://pith.science/api/pith-number/ACKR4SU34PYA43YCRFKVLXQXXG/graph.json","events_json":"https://pith.science/api/pith-number/ACKR4SU34PYA43YCRFKVLXQXXG/events.json","paper":"https://pith.science/paper/ACKR4SU3"},"agent_actions":{"view_html":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG","download_json":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG.json","view_paper":"https://pith.science/paper/ACKR4SU3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01061&json=true","fetch_graph":"https://pith.science/api/pith-number/ACKR4SU34PYA43YCRFKVLXQXXG/graph.json","fetch_events":"https://pith.science/api/pith-number/ACKR4SU34PYA43YCRFKVLXQXXG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG/action/storage_attestation","attest_author":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG/action/author_attestation","sign_citation":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG/action/citation_signature","submit_replication":"https://pith.science/pith/ACKR4SU34PYA43YCRFKVLXQXXG/action/replication_record"}},"created_at":"2026-07-05T11:29:06.426467+00:00","updated_at":"2026-07-05T11:29:06.426467+00:00"}