{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M55BPJF3FTQUGEOWMCE6OV7P5K","short_pith_number":"pith:M55BPJF3","schema_version":"1.0","canonical_sha256":"677a17a4bb2ce14311d66089e757efea968d7e6d53ee13788ed1c4c4c234d3d8","source":{"kind":"arxiv","id":"2403.14468","version":4},"attestation_state":"computed","paper":{"title":"AnyV2V: A Tuning-Free Framework For Any Video-to-Video Editing Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Cong Wei, Harry Yang, Max Ku, Weiming Ren, Wenhu Chen","submitted_at":"2024-03-21T15:15:00Z","abstract_excerpt":"In the dynamic field of digital content creation using generative models, state-of-the-art video editing models still do not offer the level of quality and control that users desire. Previous works on video editing either extended from image-based generative models in a zero-shot manner or necessitated extensive fine-tuning, which can hinder the production of fluid video edits. Furthermore, these methods frequently rely on textual input as the editing guidance, leading to ambiguities and limiting the types of edits they can perform. Recognizing these challenges, we introduce AnyV2V, a novel tu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14468","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-21T15:15:00Z","cross_cats_sorted":["cs.AI","cs.MM"],"title_canon_sha256":"7b2840a8a5879c0b9aef5956f175a7897bbff70f43554893d515fbec3f74b720","abstract_canon_sha256":"f4d127f964ab98919d8aeafaadd7efd21236b71c3cd1edb7ac5563a7b9294fae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:11.208739Z","signature_b64":"XSrI3ZDP1WzjuYoT5jqf8Z6pwyz47TGsPEMO7vbrpK5oPHJsAY8EB7Uc0zV8R8102qDNYwrvXiA7pkP87ohVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"677a17a4bb2ce14311d66089e757efea968d7e6d53ee13788ed1c4c4c234d3d8","last_reissued_at":"2026-07-05T09:30:11.208263Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:11.208263Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AnyV2V: A Tuning-Free Framework For Any Video-to-Video Editing Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Cong Wei, Harry Yang, Max Ku, Weiming Ren, Wenhu Chen","submitted_at":"2024-03-21T15:15:00Z","abstract_excerpt":"In the dynamic field of digital content creation using generative models, state-of-the-art video editing models still do not offer the level of quality and control that users desire. Previous works on video editing either extended from image-based generative models in a zero-shot manner or necessitated extensive fine-tuning, which can hinder the production of fluid video edits. Furthermore, these methods frequently rely on textual input as the editing guidance, leading to ambiguities and limiting the types of edits they can perform. Recognizing these challenges, we introduce AnyV2V, a novel tu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14468","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14468/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14468","created_at":"2026-07-05T09:30:11.208332+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14468v4","created_at":"2026-07-05T09:30:11.208332+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14468","created_at":"2026-07-05T09:30:11.208332+00:00"},{"alias_kind":"pith_short_12","alias_value":"M55BPJF3FTQU","created_at":"2026-07-05T09:30:11.208332+00:00"},{"alias_kind":"pith_short_16","alias_value":"M55BPJF3FTQUGEOW","created_at":"2026-07-05T09:30:11.208332+00:00"},{"alias_kind":"pith_short_8","alias_value":"M55BPJF3","created_at":"2026-07-05T09:30:11.208332+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23254","citing_title":"SteerVTE: Seamless Video Text Editing with Style and Glyph Control","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22042","citing_title":"IDAG-Edit: Multi-Object Video Editing via Instance-Decoupled Attention and Guidance","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01362","citing_title":"AlbedoEdit: Unified Instance-Level Video Editing with Albedo Guidance","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01399","citing_title":"PAI-Studio: Cinematic Video Background Replacement with Camera-Aware Motion","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12271","citing_title":"Beyond Text Prompts: Visual-to-Visual Generation as A Unified Paradigm","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21466","citing_title":"StreamEdit: Training-Free Video Editing via Few-Step Streaming Video Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30599","citing_title":"Goku: A Million-Scale Universal Dataset and Benchmark for Instruction-Based Video Editing","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04569","citing_title":"LIVEditor-14B: Lightning Unified Video Editing via In-Context Sparse Attention","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24674","citing_title":"Reasoning to Align: Implicit Reasoning in Diffusion Transformers for Video Editing","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25193","citing_title":"SpongeBob: Sync-Aware Harmonious Audio-Visual Generative Editing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29020","citing_title":"Semantic-Aware, Physics-Informed, Geometry-Grounded Weather Video Synthesis","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30599","citing_title":"Goku: A Million-Scale Universal Dataset and Benchmark for Instruction-Based Video Editing","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23192","citing_title":"Occlusion-Aware Physics-Semantic Keyframe Selection for Robust Video Editing","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23245","citing_title":"SimInsert: Seamless Video Object Insertion via Regional Sparse Attention Fusion","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18678","citing_title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21466","citing_title":"StreamEdit: Training-Free Video Editing via Few-Step Streaming Video Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17312","citing_title":"VISTA: Triplet-Supervised Video Style Transfer with Diffusion Transformers","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18678","citing_title":"Lance: Unified Multimodal Modeling by Multi-Task Synergy","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17019","citing_title":"StreamingEffect: Real-Time Human-Centric Video Effect Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2509.04434","citing_title":"Durian: Dual Reference Image-Guided Portrait Animation with Attribute Transfer","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01186","citing_title":"ASTRA: Let Arbitrary Subjects Transform in Video Editing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14136","citing_title":"TeDiO: Temporal Diagonal Optimization for Training-Free Coherent Video Diffusion","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12271","citing_title":"Beyond Text Prompts: Visual-to-Visual Generation as A Unified Paradigm","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21921","citing_title":"Context Unrolling in Omni Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19741","citing_title":"CityRAG: Stepping Into a City via Spatially-Grounded Video Generation","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K","json":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K.json","graph_json":"https://pith.science/api/pith-number/M55BPJF3FTQUGEOWMCE6OV7P5K/graph.json","events_json":"https://pith.science/api/pith-number/M55BPJF3FTQUGEOWMCE6OV7P5K/events.json","paper":"https://pith.science/paper/M55BPJF3"},"agent_actions":{"view_html":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K","download_json":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K.json","view_paper":"https://pith.science/paper/M55BPJF3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14468&json=true","fetch_graph":"https://pith.science/api/pith-number/M55BPJF3FTQUGEOWMCE6OV7P5K/graph.json","fetch_events":"https://pith.science/api/pith-number/M55BPJF3FTQUGEOWMCE6OV7P5K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K/action/storage_attestation","attest_author":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K/action/author_attestation","sign_citation":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K/action/citation_signature","submit_replication":"https://pith.science/pith/M55BPJF3FTQUGEOWMCE6OV7P5K/action/replication_record"}},"created_at":"2026-07-05T09:30:11.208332+00:00","updated_at":"2026-07-05T09:30:11.208332+00:00"}