{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Q6ICGRNCA7QS7YPIMEXFV3K3PV","short_pith_number":"pith:Q6ICGRNC","schema_version":"1.0","canonical_sha256":"87902345a207e12fe1e8612e5aed5b7d4580c801c00fa441e108ed41fd39babd","source":{"kind":"arxiv","id":"2112.02815","version":2},"attestation_state":"computed","paper":{"title":"Make It Move: Controllable Image-to-Video Generation with Text Descriptions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chong Luo, Yaosi Hu, Zhenzhong Chen","submitted_at":"2021-12-06T07:00:36Z","abstract_excerpt":"Generating controllable videos conforming to user intentions is an appealing yet challenging topic in computer vision. To enable maneuverable control in line with user intentions, a novel video generation task, named Text-Image-to-Video generation (TI2V), is proposed. With both controllable appearance and motion, TI2V aims at generating videos from a static image and a text description. The key challenges of TI2V task lie both in aligning appearance and motion from different modalities, and in handling uncertainty in text descriptions. To address these challenges, we propose a Motion Anchor-ba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.02815","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2021-12-06T07:00:36Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d2a1744524558ee814f633b101d66c1179bd6a4e845c6715eb52ca2ef1fe7d93","abstract_canon_sha256":"75025eb07f6379ea9e6837bd97ec12d9b2eaa08f7ef97a2317494cb6c17b57ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:24.872113Z","signature_b64":"gGpMCnAoNUgo06KlDvYiSSJyR6F+nupbp81A10FTMLzqAtL2o4WaZ9M4OH+Rlf21Jl8PbSle0KMO2LLemZKFCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"87902345a207e12fe1e8612e5aed5b7d4580c801c00fa441e108ed41fd39babd","last_reissued_at":"2026-07-05T04:10:24.871643Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:24.871643Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Make It Move: Controllable Image-to-Video Generation with Text Descriptions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chong Luo, Yaosi Hu, Zhenzhong Chen","submitted_at":"2021-12-06T07:00:36Z","abstract_excerpt":"Generating controllable videos conforming to user intentions is an appealing yet challenging topic in computer vision. To enable maneuverable control in line with user intentions, a novel video generation task, named Text-Image-to-Video generation (TI2V), is proposed. With both controllable appearance and motion, TI2V aims at generating videos from a static image and a text description. The key challenges of TI2V task lie both in aligning appearance and motion from different modalities, and in handling uncertainty in text descriptions. To address these challenges, we propose a Motion Anchor-ba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.02815","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.02815/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.02815","created_at":"2026-07-05T04:10:24.871701+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.02815v2","created_at":"2026-07-05T04:10:24.871701+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.02815","created_at":"2026-07-05T04:10:24.871701+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q6ICGRNCA7QS","created_at":"2026-07-05T04:10:24.871701+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q6ICGRNCA7QS7YPI","created_at":"2026-07-05T04:10:24.871701+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q6ICGRNC","created_at":"2026-07-05T04:10:24.871701+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.17180","citing_title":"We'll Fix it in Post: Improving Text-to-Video Generation with Neuro-Symbolic Feedback","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV","json":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV.json","graph_json":"https://pith.science/api/pith-number/Q6ICGRNCA7QS7YPIMEXFV3K3PV/graph.json","events_json":"https://pith.science/api/pith-number/Q6ICGRNCA7QS7YPIMEXFV3K3PV/events.json","paper":"https://pith.science/paper/Q6ICGRNC"},"agent_actions":{"view_html":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV","download_json":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV.json","view_paper":"https://pith.science/paper/Q6ICGRNC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.02815&json=true","fetch_graph":"https://pith.science/api/pith-number/Q6ICGRNCA7QS7YPIMEXFV3K3PV/graph.json","fetch_events":"https://pith.science/api/pith-number/Q6ICGRNCA7QS7YPIMEXFV3K3PV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV/action/storage_attestation","attest_author":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV/action/author_attestation","sign_citation":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV/action/citation_signature","submit_replication":"https://pith.science/pith/Q6ICGRNCA7QS7YPIMEXFV3K3PV/action/replication_record"}},"created_at":"2026-07-05T04:10:24.871701+00:00","updated_at":"2026-07-05T04:10:24.871701+00:00"}