{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WJCZNOSIEUXBNXHX577OA6U44O","short_pith_number":"pith:WJCZNOSI","schema_version":"1.0","canonical_sha256":"b24596ba48252e16dcf7effee07a9ce39bed7fce580797820ee7eb1ac374875b","source":{"kind":"arxiv","id":"2504.02436","version":1},"attestation_state":"computed","paper":{"title":"SkyReels-A2: Compose Anything in Video Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Debang Li, Di Qiu, Guibin Chen, Jiahua Wang, Jingtao Xu, Mingyuan Fan, Rui Wang, Yahui Zhou, Yang Li, Yikun Dou, Zhengcong Fei","submitted_at":"2025-04-03T09:50:50Z","abstract_excerpt":"This paper presents SkyReels-A2, a controllable video generation framework capable of assembling arbitrary visual elements (e.g., characters, objects, backgrounds) into synthesized videos based on textual prompts while maintaining strict consistency with reference images for each element. We term this task elements-to-video (E2V), whose primary challenges lie in preserving the fidelity of each reference element, ensuring coherent composition of the scene, and achieving natural outputs. To address these, we first design a comprehensive data pipeline to construct prompt-reference-video triplets "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.02436","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-03T09:50:50Z","cross_cats_sorted":[],"title_canon_sha256":"e341d8879736cecb00ca1ebba3bb6ae42702a2145aeab8e88fe6493aaf7950cb","abstract_canon_sha256":"96b6bab736464b64c099dde46df984ba1a4ba9b84ca4c473836806df790c9ef4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:57.026929Z","signature_b64":"n8b/a62sK23O2X/Rdcn5BbhxDMYOlNyggLa2oRO8uJiFdneoxOEB964rsoRmV+tn9GIkzGafboNNTbJ355TeDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b24596ba48252e16dcf7effee07a9ce39bed7fce580797820ee7eb1ac374875b","last_reissued_at":"2026-07-05T10:43:57.026385Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:57.026385Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SkyReels-A2: Compose Anything in Video Diffusion Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Debang Li, Di Qiu, Guibin Chen, Jiahua Wang, Jingtao Xu, Mingyuan Fan, Rui Wang, Yahui Zhou, Yang Li, Yikun Dou, Zhengcong Fei","submitted_at":"2025-04-03T09:50:50Z","abstract_excerpt":"This paper presents SkyReels-A2, a controllable video generation framework capable of assembling arbitrary visual elements (e.g., characters, objects, backgrounds) into synthesized videos based on textual prompts while maintaining strict consistency with reference images for each element. We term this task elements-to-video (E2V), whose primary challenges lie in preserving the fidelity of each reference element, ensuring coherent composition of the scene, and achieving natural outputs. To address these, we first design a comprehensive data pipeline to construct prompt-reference-video triplets "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.02436","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.02436/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.02436","created_at":"2026-07-05T10:43:57.026449+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.02436v1","created_at":"2026-07-05T10:43:57.026449+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.02436","created_at":"2026-07-05T10:43:57.026449+00:00"},{"alias_kind":"pith_short_12","alias_value":"WJCZNOSIEUXB","created_at":"2026-07-05T10:43:57.026449+00:00"},{"alias_kind":"pith_short_16","alias_value":"WJCZNOSIEUXBNXHX","created_at":"2026-07-05T10:43:57.026449+00:00"},{"alias_kind":"pith_short_8","alias_value":"WJCZNOSI","created_at":"2026-07-05T10:43:57.026449+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26058","citing_title":"DomainShuttle: Freeform Open Domain Subject-driven Text-to-video Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24107","citing_title":"DramaDirector: Geometry-Guided Short Drama Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13768","citing_title":"CineOrchestra: Unified Entity-Centric Conditioning for Cinematic Video Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11670","citing_title":"ARGUS: Stacked Multi-View Identity Mosaic Injection for Subject-Preserving Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07508","citing_title":"Streaming Video Generation with Streaming Force Control","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02441","citing_title":"Spatial-Temporal Decoupled Reference Conditioning for Identity-Preserving Text-to-Video Generation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15824","citing_title":"FashionChameleon: Towards Real-Time and Interactive Human-Garment Video Customization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22344","citing_title":"Bernini: Latent Semantic Planning for Video Diffusion","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22051","citing_title":"EasyVFX: Frequency-Driven Decoupling for Resource-Efficient VFX Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15824","citing_title":"FashionChameleon: Towards Real-Time and Interactive Human-Garment Video Customization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17488","citing_title":"Omni-Customizer: End-to-End MultiModal Customization for Joint Audio-Video Generation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13074","citing_title":"SkyReels-V2: Infinite-length Film Generative Model","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27918","citing_title":"Generate Your Talking Avatar from Video Reference","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04702","citing_title":"FaithfulFaces: Pose-Faithful Facial Identity Preservation for Text-to-Video Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02948","citing_title":"AsymTalker: Identity-Consistent Long-Term Talking Head Generation via Asymmetric Distillation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19679","citing_title":"MMControl: Unified Multi-Modal Control for Joint Audio-Video Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11804","citing_title":"OmniShow: Unifying Multimodal Conditions for Human-Object Interaction Video Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02948","citing_title":"AsymTalker: Identity-Consistent Long-Term Talking Head Generation via Asymmetric Distillation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":203,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O","json":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O.json","graph_json":"https://pith.science/api/pith-number/WJCZNOSIEUXBNXHX577OA6U44O/graph.json","events_json":"https://pith.science/api/pith-number/WJCZNOSIEUXBNXHX577OA6U44O/events.json","paper":"https://pith.science/paper/WJCZNOSI"},"agent_actions":{"view_html":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O","download_json":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O.json","view_paper":"https://pith.science/paper/WJCZNOSI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.02436&json=true","fetch_graph":"https://pith.science/api/pith-number/WJCZNOSIEUXBNXHX577OA6U44O/graph.json","fetch_events":"https://pith.science/api/pith-number/WJCZNOSIEUXBNXHX577OA6U44O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O/action/storage_attestation","attest_author":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O/action/author_attestation","sign_citation":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O/action/citation_signature","submit_replication":"https://pith.science/pith/WJCZNOSIEUXBNXHX577OA6U44O/action/replication_record"}},"created_at":"2026-07-05T10:43:57.026449+00:00","updated_at":"2026-07-05T10:43:57.026449+00:00"}