{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WEAA53AMSMMRDZALMV6YOXJVWY","short_pith_number":"pith:WEAA53AM","schema_version":"1.0","canonical_sha256":"b1000eec0c931911e40b657d875d35b62907c8f060ad39057b9da5aa908bdd3f","source":{"kind":"arxiv","id":"2402.03162","version":2},"attestation_state":"computed","paper":{"title":"Direct-a-Video: Customized Video Generation with User-Directed Camera Movement and Object Motion","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongyang Ma, Di Zhang, Haibin Huang, Jing Liao, Liang Hou, Pengfei Wan, Shiyuan Yang, Xiaodong Chen","submitted_at":"2024-02-05T16:30:57Z","abstract_excerpt":"Recent text-to-video diffusion models have achieved impressive progress. In practice, users often desire the ability to control object motion and camera movement independently for customized video creation. However, current methods lack the focus on separately controlling object motion and camera movement in a decoupled manner, which limits the controllability and flexibility of text-to-video models. In this paper, we introduce Direct-a-Video, a system that allows users to independently specify motions for multiple objects as well as camera's pan and zoom movements, as if directing a video. We"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.03162","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-05T16:30:57Z","cross_cats_sorted":[],"title_canon_sha256":"4f31331c5f1d304efb94af0a423238ca02623886ea2e591c48732c13e5f8b247","abstract_canon_sha256":"b7cbc95ec6d118eec017f68f76d6075174cc30c35e4cb63c46c1c282858f416a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:45.029827Z","signature_b64":"MUX9d+l9UhEiZfu8+fS35edUDHWkYIWZ4hjKPw+Mzf2ZVmi6GIi4fXXpO0vHfkjp4Qd/YGur3ADgq5/rINIyBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b1000eec0c931911e40b657d875d35b62907c8f060ad39057b9da5aa908bdd3f","last_reissued_at":"2026-07-05T08:15:45.029392Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:45.029392Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct-a-Video: Customized Video Generation with User-Directed Camera Movement and Object Motion","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongyang Ma, Di Zhang, Haibin Huang, Jing Liao, Liang Hou, Pengfei Wan, Shiyuan Yang, Xiaodong Chen","submitted_at":"2024-02-05T16:30:57Z","abstract_excerpt":"Recent text-to-video diffusion models have achieved impressive progress. In practice, users often desire the ability to control object motion and camera movement independently for customized video creation. However, current methods lack the focus on separately controlling object motion and camera movement in a decoupled manner, which limits the controllability and flexibility of text-to-video models. In this paper, we introduce Direct-a-Video, a system that allows users to independently specify motions for multiple objects as well as camera's pan and zoom movements, as if directing a video. We"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.03162","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.03162/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.03162","created_at":"2026-07-05T08:15:45.029460+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.03162v2","created_at":"2026-07-05T08:15:45.029460+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.03162","created_at":"2026-07-05T08:15:45.029460+00:00"},{"alias_kind":"pith_short_12","alias_value":"WEAA53AMSMMR","created_at":"2026-07-05T08:15:45.029460+00:00"},{"alias_kind":"pith_short_16","alias_value":"WEAA53AMSMMRDZAL","created_at":"2026-07-05T08:15:45.029460+00:00"},{"alias_kind":"pith_short_8","alias_value":"WEAA53AM","created_at":"2026-07-05T08:15:45.029460+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.18576","citing_title":"DriVerse: Navigation World Model for Driving Simulation via Multimodal Trajectory Prompting and Motion Alignment","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2404.02101","citing_title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","ref_index":160,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY","json":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY.json","graph_json":"https://pith.science/api/pith-number/WEAA53AMSMMRDZALMV6YOXJVWY/graph.json","events_json":"https://pith.science/api/pith-number/WEAA53AMSMMRDZALMV6YOXJVWY/events.json","paper":"https://pith.science/paper/WEAA53AM"},"agent_actions":{"view_html":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY","download_json":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY.json","view_paper":"https://pith.science/paper/WEAA53AM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.03162&json=true","fetch_graph":"https://pith.science/api/pith-number/WEAA53AMSMMRDZALMV6YOXJVWY/graph.json","fetch_events":"https://pith.science/api/pith-number/WEAA53AMSMMRDZALMV6YOXJVWY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY/action/storage_attestation","attest_author":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY/action/author_attestation","sign_citation":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY/action/citation_signature","submit_replication":"https://pith.science/pith/WEAA53AMSMMRDZALMV6YOXJVWY/action/replication_record"}},"created_at":"2026-07-05T08:15:45.029460+00:00","updated_at":"2026-07-05T08:15:45.029460+00:00"}