{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZQJFYQDJQWIRXOQAAVLASU3LU3","short_pith_number":"pith:ZQJFYQDJ","schema_version":"1.0","canonical_sha256":"cc125c406985911bba00055609536ba6fc1858980fb29e114870dcd664769d3c","source":{"kind":"arxiv","id":"2408.13005","version":2},"attestation_state":"computed","paper":{"title":"EasyControl: Transfer ControlNet to Video Diffusion for Controllable Generation and Interpolation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wang, Hang Xu, Haoyu Zhao, Jianhua Han, Jiaxi Gu, Panwen Hu, Xiaodan Liang, Yuanfan Guo","submitted_at":"2024-08-23T11:48:29Z","abstract_excerpt":"Following the advancements in text-guided image generation technology exemplified by Stable Diffusion, video generation is gaining increased attention in the academic community. However, relying solely on text guidance for video generation has serious limitations, as videos contain much richer content than images, especially in terms of motion. This information can hardly be adequately described with plain text. Fortunately, in computer vision, various visual representations can serve as additional control signals to guide generation. With the help of these signals, video generation can be con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.13005","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-23T11:48:29Z","cross_cats_sorted":[],"title_canon_sha256":"49b9314dcffa70a0fb6320bb320508c0291782d82117de168557e71158f8201d","abstract_canon_sha256":"0f552a0df8fba1ef57544ce22163c8f908e1bee013403d435da0a1292a0930a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:31.132770Z","signature_b64":"ngfS841zXFiufYgNPot0atsvSOpymAxUE340BiVPxJtth6BXa4wZ6snXDNyoxn9BhZ5f3yi7mLoze34DwSaoBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc125c406985911bba00055609536ba6fc1858980fb29e114870dcd664769d3c","last_reissued_at":"2026-07-05T09:07:31.132303Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:31.132303Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EasyControl: Transfer ControlNet to Video Diffusion for Controllable Generation and Interpolation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cong Wang, Hang Xu, Haoyu Zhao, Jianhua Han, Jiaxi Gu, Panwen Hu, Xiaodan Liang, Yuanfan Guo","submitted_at":"2024-08-23T11:48:29Z","abstract_excerpt":"Following the advancements in text-guided image generation technology exemplified by Stable Diffusion, video generation is gaining increased attention in the academic community. However, relying solely on text guidance for video generation has serious limitations, as videos contain much richer content than images, especially in terms of motion. This information can hardly be adequately described with plain text. Fortunately, in computer vision, various visual representations can serve as additional control signals to guide generation. With the help of these signals, video generation can be con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.13005","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.13005/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.13005","created_at":"2026-07-05T09:07:31.132369+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.13005v2","created_at":"2026-07-05T09:07:31.132369+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.13005","created_at":"2026-07-05T09:07:31.132369+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZQJFYQDJQWIR","created_at":"2026-07-05T09:07:31.132369+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZQJFYQDJQWIRXOQA","created_at":"2026-07-05T09:07:31.132369+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZQJFYQDJ","created_at":"2026-07-05T09:07:31.132369+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08770","citing_title":"LongE2V: Long-Horizon Event-based Video Reconstruction, Prediction, and Frame Interpolation with Video Diffusion Models","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2603.20725","citing_title":"Premier: Personalized Preference Modulation with Learnable User Embedding in Text-to-Image Generation","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3","json":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3.json","graph_json":"https://pith.science/api/pith-number/ZQJFYQDJQWIRXOQAAVLASU3LU3/graph.json","events_json":"https://pith.science/api/pith-number/ZQJFYQDJQWIRXOQAAVLASU3LU3/events.json","paper":"https://pith.science/paper/ZQJFYQDJ"},"agent_actions":{"view_html":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3","download_json":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3.json","view_paper":"https://pith.science/paper/ZQJFYQDJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.13005&json=true","fetch_graph":"https://pith.science/api/pith-number/ZQJFYQDJQWIRXOQAAVLASU3LU3/graph.json","fetch_events":"https://pith.science/api/pith-number/ZQJFYQDJQWIRXOQAAVLASU3LU3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3/action/storage_attestation","attest_author":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3/action/author_attestation","sign_citation":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3/action/citation_signature","submit_replication":"https://pith.science/pith/ZQJFYQDJQWIRXOQAAVLASU3LU3/action/replication_record"}},"created_at":"2026-07-05T09:07:31.132369+00:00","updated_at":"2026-07-05T09:07:31.132369+00:00"}