{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3LADNHXEPBRAW4ON6XKYUTPKUT","short_pith_number":"pith:3LADNHXE","schema_version":"1.0","canonical_sha256":"dac0369ee478620b71cdf5d58a4deaa4fc23213b51fbb33197f739edc3d64710","source":{"kind":"arxiv","id":"2308.14749","version":1},"attestation_state":"computed","paper":{"title":"MagicEdit: High-Fidelity and Temporally Coherent Video Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hanshu Yan, Jianfeng Zhang, Jiashi Feng, Jun Hao Liew, Zhongcong Xu","submitted_at":"2023-08-28T17:56:22Z","abstract_excerpt":"In this report, we present MagicEdit, a surprisingly simple yet effective solution to the text-guided video editing task. We found that high-fidelity and temporally coherent video-to-video translation can be achieved by explicitly disentangling the learning of content, structure and motion signals during training. This is in contradict to most existing methods which attempt to jointly model both the appearance and temporal representation within a single framework, which we argue, would lead to degradation in per-frame quality. Despite its simplicity, we show that MagicEdit supports various dow"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.14749","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-28T17:56:22Z","cross_cats_sorted":[],"title_canon_sha256":"3ebb81848bc9bd9ae88fdd309f7d4cacdf83c753f657e84fb5641fb7eba87971","abstract_canon_sha256":"1e8a1abd4f89dd9ede9df82caa4e3bd2e61bab08407f666b2a0dd7794b763c0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:45:30.635716Z","signature_b64":"NUxVaVx2jATfs8QCaB22wUsWH+Te68oDdKJLx1GowRvKMIOM3DYIAOVB2NN99xhESYrh/B3oN6c2N+CM+M5YBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dac0369ee478620b71cdf5d58a4deaa4fc23213b51fbb33197f739edc3d64710","last_reissued_at":"2026-07-05T06:45:30.635217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:45:30.635217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MagicEdit: High-Fidelity and Temporally Coherent Video Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hanshu Yan, Jianfeng Zhang, Jiashi Feng, Jun Hao Liew, Zhongcong Xu","submitted_at":"2023-08-28T17:56:22Z","abstract_excerpt":"In this report, we present MagicEdit, a surprisingly simple yet effective solution to the text-guided video editing task. We found that high-fidelity and temporally coherent video-to-video translation can be achieved by explicitly disentangling the learning of content, structure and motion signals during training. This is in contradict to most existing methods which attempt to jointly model both the appearance and temporal representation within a single framework, which we argue, would lead to degradation in per-frame quality. Despite its simplicity, we show that MagicEdit supports various dow"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.14749","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.14749/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.14749","created_at":"2026-07-05T06:45:30.635275+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.14749v1","created_at":"2026-07-05T06:45:30.635275+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.14749","created_at":"2026-07-05T06:45:30.635275+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LADNHXEPBRA","created_at":"2026-07-05T06:45:30.635275+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LADNHXEPBRAW4ON","created_at":"2026-07-05T06:45:30.635275+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LADNHXE","created_at":"2026-07-05T06:45:30.635275+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.07598","citing_title":"VACE: All-in-One Video Creation and Editing","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14125","citing_title":"VideoPoet: A Large Language Model for Zero-Shot Video Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21755","citing_title":"VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2406.09414","citing_title":"Depth Anything V2","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":260,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT","json":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT.json","graph_json":"https://pith.science/api/pith-number/3LADNHXEPBRAW4ON6XKYUTPKUT/graph.json","events_json":"https://pith.science/api/pith-number/3LADNHXEPBRAW4ON6XKYUTPKUT/events.json","paper":"https://pith.science/paper/3LADNHXE"},"agent_actions":{"view_html":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT","download_json":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT.json","view_paper":"https://pith.science/paper/3LADNHXE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.14749&json=true","fetch_graph":"https://pith.science/api/pith-number/3LADNHXEPBRAW4ON6XKYUTPKUT/graph.json","fetch_events":"https://pith.science/api/pith-number/3LADNHXEPBRAW4ON6XKYUTPKUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT/action/storage_attestation","attest_author":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT/action/author_attestation","sign_citation":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT/action/citation_signature","submit_replication":"https://pith.science/pith/3LADNHXEPBRAW4ON6XKYUTPKUT/action/replication_record"}},"created_at":"2026-07-05T06:45:30.635275+00:00","updated_at":"2026-07-05T06:45:30.635275+00:00"}