{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:YWLWT6WH34LS2LS7FUBABX6QMA","short_pith_number":"pith:YWLWT6WH","schema_version":"1.0","canonical_sha256":"c59769fac7df172d2e5f2d0200dfd0602bdde4d2447fcf3a27ee00c677f96811","source":{"kind":"arxiv","id":"2203.02573","version":1},"attestation_state":"computed","paper":{"title":"Show Me What and Tell Me How: Video Synthesis via Multimodal Conditioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dimitris Metaxas, Francesco Barbieri, Hsin-Ying Lee, Jian Ren, Kyle Olszewski, Ligong Han, Sergey Tulyakov, Shervin Minaee","submitted_at":"2022-03-04T21:09:13Z","abstract_excerpt":"Most methods for conditional video synthesis use a single modality as the condition. This comes with major limitations. For example, it is problematic for a model conditioned on an image to generate a specific motion trajectory desired by the user since there is no means to provide motion information. Conversely, language information can describe the desired motion, while not precisely defining the content of the video. This work presents a multimodal video generation framework that benefits from text and images provided jointly or separately. We leverage the recent progress in quantized repre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.02573","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-04T21:09:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f70f62cba77b4a154d253e0130150b87f240c9847c037971ebbe70c6c51cc758","abstract_canon_sha256":"7a8d7bfa59ea539a63ae9447a90bd75e50fe0eadf43c40ef4c79ca8743d39003"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:02:24.714727Z","signature_b64":"JGHxxYrQsKyc+nELKExUBeAz9lCXEEprxeDtK0an4WvgDn0NfOhPIsCVKsWgql+li1HccFITV7v/VvGmmMuCAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c59769fac7df172d2e5f2d0200dfd0602bdde4d2447fcf3a27ee00c677f96811","last_reissued_at":"2026-07-05T04:02:24.714243Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:02:24.714243Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Show Me What and Tell Me How: Video Synthesis via Multimodal Conditioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Dimitris Metaxas, Francesco Barbieri, Hsin-Ying Lee, Jian Ren, Kyle Olszewski, Ligong Han, Sergey Tulyakov, Shervin Minaee","submitted_at":"2022-03-04T21:09:13Z","abstract_excerpt":"Most methods for conditional video synthesis use a single modality as the condition. This comes with major limitations. For example, it is problematic for a model conditioned on an image to generate a specific motion trajectory desired by the user since there is no means to provide motion information. Conversely, language information can describe the desired motion, while not precisely defining the content of the video. This work presents a multimodal video generation framework that benefits from text and images provided jointly or separately. We leverage the recent progress in quantized repre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.02573","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.02573/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.02573","created_at":"2026-07-05T04:02:24.714302+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.02573v1","created_at":"2026-07-05T04:02:24.714302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.02573","created_at":"2026-07-05T04:02:24.714302+00:00"},{"alias_kind":"pith_short_12","alias_value":"YWLWT6WH34LS","created_at":"2026-07-05T04:02:24.714302+00:00"},{"alias_kind":"pith_short_16","alias_value":"YWLWT6WH34LS2LS7","created_at":"2026-07-05T04:02:24.714302+00:00"},{"alias_kind":"pith_short_8","alias_value":"YWLWT6WH","created_at":"2026-07-05T04:02:24.714302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA","json":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA.json","graph_json":"https://pith.science/api/pith-number/YWLWT6WH34LS2LS7FUBABX6QMA/graph.json","events_json":"https://pith.science/api/pith-number/YWLWT6WH34LS2LS7FUBABX6QMA/events.json","paper":"https://pith.science/paper/YWLWT6WH"},"agent_actions":{"view_html":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA","download_json":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA.json","view_paper":"https://pith.science/paper/YWLWT6WH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.02573&json=true","fetch_graph":"https://pith.science/api/pith-number/YWLWT6WH34LS2LS7FUBABX6QMA/graph.json","fetch_events":"https://pith.science/api/pith-number/YWLWT6WH34LS2LS7FUBABX6QMA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA/action/storage_attestation","attest_author":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA/action/author_attestation","sign_citation":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA/action/citation_signature","submit_replication":"https://pith.science/pith/YWLWT6WH34LS2LS7FUBABX6QMA/action/replication_record"}},"created_at":"2026-07-05T04:02:24.714302+00:00","updated_at":"2026-07-05T04:02:24.714302+00:00"}