{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:F4OJUXDINNSJKM2GTSYWS7CM2V","short_pith_number":"pith:F4OJUXDI","schema_version":"1.0","canonical_sha256":"2f1c9a5c686b649533469cb1697c4cd57c7b2af8c9a0f2336b18ea9b58e94669","source":{"kind":"arxiv","id":"2302.01329","version":1},"attestation_state":"computed","paper":{"title":"Dreamix: Video Diffusion Models are General Video Editors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Rav Acha, Dani Valevski, Eliahu Horwitz, Eyal Molad, Yael Pritch, Yaniv Leviathan, Yedid Hoshen, Yossi Matias","submitted_at":"2023-02-02T18:58:58Z","abstract_excerpt":"Text-driven image and video diffusion models have recently achieved unprecedented generation realism. While diffusion models have been successfully applied for image editing, very few works have done so for video editing. We present the first diffusion-based method that is able to perform text-based motion and appearance editing of general videos. Our approach uses a video diffusion model to combine, at inference time, the low-resolution spatio-temporal information from the original video with new, high resolution information that it synthesized to align with the guiding text prompt. As obtain"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.01329","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-02-02T18:58:58Z","cross_cats_sorted":[],"title_canon_sha256":"02e81e5e4cfd173e4e198a3fd8b0b16b30401095fce17387766a158c750a4ad5","abstract_canon_sha256":"ae278ef4f31cd43c3e1ed6a3d9f640e2d9a569d68049360f4f3efd5c05e397a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:38:25.086351Z","signature_b64":"ZNtBRg8KQxHZO7r2JXChioHLTcb5Jpb/KEiZUhz22DCzPRIvPtplKmkgkygPRpmd+2dblDDd/oG/P+doXGU/AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f1c9a5c686b649533469cb1697c4cd57c7b2af8c9a0f2336b18ea9b58e94669","last_reissued_at":"2026-07-05T05:38:25.085909Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:38:25.085909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dreamix: Video Diffusion Models are General Video Editors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Rav Acha, Dani Valevski, Eliahu Horwitz, Eyal Molad, Yael Pritch, Yaniv Leviathan, Yedid Hoshen, Yossi Matias","submitted_at":"2023-02-02T18:58:58Z","abstract_excerpt":"Text-driven image and video diffusion models have recently achieved unprecedented generation realism. While diffusion models have been successfully applied for image editing, very few works have done so for video editing. We present the first diffusion-based method that is able to perform text-based motion and appearance editing of general videos. Our approach uses a video diffusion model to combine, at inference time, the low-resolution spatio-temporal information from the original video with new, high resolution information that it synthesized to align with the guiding text prompt. As obtain"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.01329","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.01329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.01329","created_at":"2026-07-05T05:38:25.085971+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.01329v1","created_at":"2026-07-05T05:38:25.085971+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.01329","created_at":"2026-07-05T05:38:25.085971+00:00"},{"alias_kind":"pith_short_12","alias_value":"F4OJUXDINNSJ","created_at":"2026-07-05T05:38:25.085971+00:00"},{"alias_kind":"pith_short_16","alias_value":"F4OJUXDINNSJKM2G","created_at":"2026-07-05T05:38:25.085971+00:00"},{"alias_kind":"pith_short_8","alias_value":"F4OJUXDI","created_at":"2026-07-05T05:38:25.085971+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19676","citing_title":"TeleMorpher: Toward Robust Simultaneous Motion-Location Editing","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2504.17180","citing_title":"We'll Fix it in Post: Improving Text-to-Video Generation with Neuro-Symbolic Feedback","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18010","citing_title":"Functionalization via Structure Completion and Motion Rectification","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18467","citing_title":"InstructAV2AV: Instruction-Guided Audio-Video Joint Editing","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2302.11550","citing_title":"Scaling Robot Learning with Semantically Imagined Experience","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":235,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06114","citing_title":"Learning Interactive Real-World Simulators","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2310.19512","citing_title":"VideoCrafter1: Open Diffusion Models for High-Quality Video Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20328","citing_title":"Video models are zero-shot learners and reasoners","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08546","citing_title":"When Numbers Speak: Aligning Textual Numerals and Visual Instances in Text-to-Video Diffusion Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13841","citing_title":"DiffMagicFace: Identity Consistent Facial Editing of Real Videos","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V","json":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V.json","graph_json":"https://pith.science/api/pith-number/F4OJUXDINNSJKM2GTSYWS7CM2V/graph.json","events_json":"https://pith.science/api/pith-number/F4OJUXDINNSJKM2GTSYWS7CM2V/events.json","paper":"https://pith.science/paper/F4OJUXDI"},"agent_actions":{"view_html":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V","download_json":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V.json","view_paper":"https://pith.science/paper/F4OJUXDI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.01329&json=true","fetch_graph":"https://pith.science/api/pith-number/F4OJUXDINNSJKM2GTSYWS7CM2V/graph.json","fetch_events":"https://pith.science/api/pith-number/F4OJUXDINNSJKM2GTSYWS7CM2V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V/action/storage_attestation","attest_author":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V/action/author_attestation","sign_citation":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V/action/citation_signature","submit_replication":"https://pith.science/pith/F4OJUXDINNSJKM2GTSYWS7CM2V/action/replication_record"}},"created_at":"2026-07-05T05:38:25.085971+00:00","updated_at":"2026-07-05T05:38:25.085971+00:00"}