{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V5B4ATCV2KMHBAZH7S6PSAWM7N","short_pith_number":"pith:V5B4ATCV","schema_version":"1.0","canonical_sha256":"af43c04c55d298708327fcbcf902ccfb6ed36eb6f5aa555514e1d88983818f5f","source":{"kind":"arxiv","id":"2401.01827","version":1},"attestation_state":"computed","paper":{"title":"Moonshot: Towards Controllable Video Generation and Editing with Multimodal Conditions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Caiming Xiong, David Junhao Zhang, Dongxu Li, Doyen Sahoo, Hung Le, Mike Zheng Shou","submitted_at":"2024-01-03T16:43:47Z","abstract_excerpt":"Most existing video diffusion models (VDMs) are limited to mere text conditions. Thereby, they are usually lacking in control over visual appearance and geometry structure of the generated videos. This work presents Moonshot, a new video generation model that conditions simultaneously on multimodal inputs of image and text. The model builts upon a core module, called multimodal video block (MVB), which consists of conventional spatialtemporal layers for representing video features, and a decoupled cross-attention layer to address image and text inputs for appearance conditioning. In addition, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.01827","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-03T16:43:47Z","cross_cats_sorted":[],"title_canon_sha256":"e229ba1c2ddcea5a36180c246ad28c06df3af1149a1858258be7e44638df0145","abstract_canon_sha256":"9cb019a265d0914f376f6b3f696f571c94f29a7783cd75e6ec8b32cfe5414408"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:57.915993Z","signature_b64":"9fOCU7cz6mfanXZumVTBBx9FtoTsnb8/ZYqKWNPvp/BIS/P3rnKfnQaxiwwdyfFSgqVbfhKpbPx1MgV7PqXUDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af43c04c55d298708327fcbcf902ccfb6ed36eb6f5aa555514e1d88983818f5f","last_reissued_at":"2026-07-05T07:29:57.915584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:57.915584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Moonshot: Towards Controllable Video Generation and Editing with Multimodal Conditions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Caiming Xiong, David Junhao Zhang, Dongxu Li, Doyen Sahoo, Hung Le, Mike Zheng Shou","submitted_at":"2024-01-03T16:43:47Z","abstract_excerpt":"Most existing video diffusion models (VDMs) are limited to mere text conditions. Thereby, they are usually lacking in control over visual appearance and geometry structure of the generated videos. This work presents Moonshot, a new video generation model that conditions simultaneously on multimodal inputs of image and text. The model builts upon a core module, called multimodal video block (MVB), which consists of conventional spatialtemporal layers for representing video features, and a decoupled cross-attention layer to address image and text inputs for appearance conditioning. In addition, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.01827","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.01827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.01827","created_at":"2026-07-05T07:29:57.915643+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.01827v1","created_at":"2026-07-05T07:29:57.915643+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.01827","created_at":"2026-07-05T07:29:57.915643+00:00"},{"alias_kind":"pith_short_12","alias_value":"V5B4ATCV2KMH","created_at":"2026-07-05T07:29:57.915643+00:00"},{"alias_kind":"pith_short_16","alias_value":"V5B4ATCV2KMHBAZH","created_at":"2026-07-05T07:29:57.915643+00:00"},{"alias_kind":"pith_short_8","alias_value":"V5B4ATCV","created_at":"2026-07-05T07:29:57.915643+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.17180","citing_title":"We'll Fix it in Post: Improving Text-to-Video Generation with Neuro-Symbolic Feedback","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17248","citing_title":"Image-to-Video Diffusion: From Foundations to Open Frontiers","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N","json":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N.json","graph_json":"https://pith.science/api/pith-number/V5B4ATCV2KMHBAZH7S6PSAWM7N/graph.json","events_json":"https://pith.science/api/pith-number/V5B4ATCV2KMHBAZH7S6PSAWM7N/events.json","paper":"https://pith.science/paper/V5B4ATCV"},"agent_actions":{"view_html":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N","download_json":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N.json","view_paper":"https://pith.science/paper/V5B4ATCV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.01827&json=true","fetch_graph":"https://pith.science/api/pith-number/V5B4ATCV2KMHBAZH7S6PSAWM7N/graph.json","fetch_events":"https://pith.science/api/pith-number/V5B4ATCV2KMHBAZH7S6PSAWM7N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N/action/storage_attestation","attest_author":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N/action/author_attestation","sign_citation":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N/action/citation_signature","submit_replication":"https://pith.science/pith/V5B4ATCV2KMHBAZH7S6PSAWM7N/action/replication_record"}},"created_at":"2026-07-05T07:29:57.915643+00:00","updated_at":"2026-07-05T07:29:57.915643+00:00"}