{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HTYDKDJBXMSN434N35EL35F2BG","short_pith_number":"pith:HTYDKDJB","schema_version":"1.0","canonical_sha256":"3cf0350d21bb24de6f8ddf48bdf4ba0982a270a2f09991532587c37b8fcf8bec","source":{"kind":"arxiv","id":"2403.14148","version":1},"attestation_state":"computed","paper":{"title":"Efficient Video Diffusion Models via Content-Frame Motion-Latent Decomposition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Anima Anandkumar, Boyi Li, De-An Huang, Jinwoo Shin, Sihyun Yu, Weili Nie","submitted_at":"2024-03-21T05:48:48Z","abstract_excerpt":"Video diffusion models have recently made great progress in generation quality, but are still limited by the high memory and computational requirements. This is because current video diffusion models often attempt to process high-dimensional videos directly. To tackle this issue, we propose content-motion latent diffusion model (CMD), a novel efficient extension of pretrained image diffusion models for video generation. Specifically, we propose an autoencoder that succinctly encodes a video as a combination of a content frame (like an image) and a low-dimensional motion latent representation. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14148","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-21T05:48:48Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6a73d4510996e1f30c3cbd4c71646082fdb7dd2eaa80030311fe1316983241df","abstract_canon_sha256":"3132f6384cc3e3b5e1919b11e6f9da0d335ff8c905cbb2c91a1609202ffbdca8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:58:55.158517Z","signature_b64":"iIGFPXKXPTNty62HWzmy4aOm2Pu0d95l2CM6WZYfcLya6wIkgvIf8obbojsImTg3ypQSZI0zbG0+fXqaKbvICQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3cf0350d21bb24de6f8ddf48bdf4ba0982a270a2f09991532587c37b8fcf8bec","last_reissued_at":"2026-07-05T07:58:55.158071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:58:55.158071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Video Diffusion Models via Content-Frame Motion-Latent Decomposition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Anima Anandkumar, Boyi Li, De-An Huang, Jinwoo Shin, Sihyun Yu, Weili Nie","submitted_at":"2024-03-21T05:48:48Z","abstract_excerpt":"Video diffusion models have recently made great progress in generation quality, but are still limited by the high memory and computational requirements. This is because current video diffusion models often attempt to process high-dimensional videos directly. To tackle this issue, we propose content-motion latent diffusion model (CMD), a novel efficient extension of pretrained image diffusion models for video generation. Specifically, we propose an autoencoder that succinctly encodes a video as a combination of a content frame (like an image) and a low-dimensional motion latent representation. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14148","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14148/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14148","created_at":"2026-07-05T07:58:55.158127+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14148v1","created_at":"2026-07-05T07:58:55.158127+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14148","created_at":"2026-07-05T07:58:55.158127+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTYDKDJBXMSN","created_at":"2026-07-05T07:58:55.158127+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTYDKDJBXMSN434N","created_at":"2026-07-05T07:58:55.158127+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTYDKDJB","created_at":"2026-07-05T07:58:55.158127+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17590","citing_title":"TivTok: Broadcasting Time-Invariant Tokens for Scalable Video Tokenization","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10183","citing_title":"Making Time Editable in Video Diffusion Transformers","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01620","citing_title":"Real-Time Generation of Streamable Talking Portrait Video with Reference-Guided Deep Compression VAEs","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31314","citing_title":"AR Forcing: Towards Long-Horizon Robot Navigation World Model","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02509","citing_title":"CamCo: Camera-Controllable 3D-Consistent Image-to-Video Generation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02134","citing_title":"Video Generation with Predictive Latents","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16479","citing_title":"Latent-Compressed Variational Autoencoder for Video Diffusion Models","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":168,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG","json":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG.json","graph_json":"https://pith.science/api/pith-number/HTYDKDJBXMSN434N35EL35F2BG/graph.json","events_json":"https://pith.science/api/pith-number/HTYDKDJBXMSN434N35EL35F2BG/events.json","paper":"https://pith.science/paper/HTYDKDJB"},"agent_actions":{"view_html":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG","download_json":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG.json","view_paper":"https://pith.science/paper/HTYDKDJB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14148&json=true","fetch_graph":"https://pith.science/api/pith-number/HTYDKDJBXMSN434N35EL35F2BG/graph.json","fetch_events":"https://pith.science/api/pith-number/HTYDKDJBXMSN434N35EL35F2BG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG/action/storage_attestation","attest_author":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG/action/author_attestation","sign_citation":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG/action/citation_signature","submit_replication":"https://pith.science/pith/HTYDKDJBXMSN434N35EL35F2BG/action/replication_record"}},"created_at":"2026-07-05T07:58:55.158127+00:00","updated_at":"2026-07-05T07:58:55.158127+00:00"}