{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YHRGUX6FP4U66NORN2UOVANQIC","short_pith_number":"pith:YHRGUX6F","schema_version":"1.0","canonical_sha256":"c1e26a5fc57f29ef35d16ea8ea81b040921e76859eb5377bb859ad71b98eafb7","source":{"kind":"arxiv","id":"2506.04590","version":1},"attestation_state":"computed","paper":{"title":"Follow-Your-Creation: Empowering 4D Creation through Video Inpainting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ayden Yang, David Junhao Zhang, Hongyu Liu, Jinbo Xing, Kunyu Feng, Qifeng Chen, Xinhua Zhang, Yinhan Zhang, Yue Ma, Zeyu Wang","submitted_at":"2025-06-05T03:11:48Z","abstract_excerpt":"We introduce Follow-Your-Creation, a novel 4D video creation framework capable of both generating and editing 4D content from a single monocular video input. By leveraging a powerful video inpainting foundation model as a generative prior, we reformulate 4D video creation as a video inpainting task, enabling the model to fill in missing content caused by camera trajectory changes or user edits. To facilitate this, we generate composite masked inpainting video data to effectively fine-tune the model for 4D video generation. Given an input video and its associated camera trajectory, we first per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04590","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-05T03:11:48Z","cross_cats_sorted":[],"title_canon_sha256":"c2ce32f4e7eef6ed9283bac985cbdf3f16db70582d86af7f1ed44d65fea382b2","abstract_canon_sha256":"a0df8017a6a63c256ccbe95acfebd0a095ce7bbde32355ea485b77d381560e78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:17.336911Z","signature_b64":"sy0Pv/UmhfS7XEDJxsMPzblibgJ5/1pGUrno40cGkPAAgu09VsJutk2BFmvlGLkhVCQAvQ5KvgV1M93vvzb9BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c1e26a5fc57f29ef35d16ea8ea81b040921e76859eb5377bb859ad71b98eafb7","last_reissued_at":"2026-07-05T11:16:17.336356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:17.336356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Follow-Your-Creation: Empowering 4D Creation through Video Inpainting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ayden Yang, David Junhao Zhang, Hongyu Liu, Jinbo Xing, Kunyu Feng, Qifeng Chen, Xinhua Zhang, Yinhan Zhang, Yue Ma, Zeyu Wang","submitted_at":"2025-06-05T03:11:48Z","abstract_excerpt":"We introduce Follow-Your-Creation, a novel 4D video creation framework capable of both generating and editing 4D content from a single monocular video input. By leveraging a powerful video inpainting foundation model as a generative prior, we reformulate 4D video creation as a video inpainting task, enabling the model to fill in missing content caused by camera trajectory changes or user edits. To facilitate this, we generate composite masked inpainting video data to effectively fine-tune the model for 4D video generation. Given an input video and its associated camera trajectory, we first per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04590","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04590/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04590","created_at":"2026-07-05T11:16:17.336429+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04590v1","created_at":"2026-07-05T11:16:17.336429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04590","created_at":"2026-07-05T11:16:17.336429+00:00"},{"alias_kind":"pith_short_12","alias_value":"YHRGUX6FP4U6","created_at":"2026-07-05T11:16:17.336429+00:00"},{"alias_kind":"pith_short_16","alias_value":"YHRGUX6FP4U66NOR","created_at":"2026-07-05T11:16:17.336429+00:00"},{"alias_kind":"pith_short_8","alias_value":"YHRGUX6F","created_at":"2026-07-05T11:16:17.336429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26740","citing_title":"LiveEdit: Towards Real-Time Diffusion-Based Streaming Video Editing","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26668","citing_title":"Disco-LoRA: Disentangled Composition of Content, Style, and Motion for Multi-concept Video Customization","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01677","citing_title":"ICDepth: Taming Video Diffusion Models for Video Depth Estimation via In-Context Conditioning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09187","citing_title":"CP4D: Compositional Physics-aware 4D Scene Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03216","citing_title":"Follow-Your-Preference++: Rethinking Preference Alignment for Image Inpainting","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30003","citing_title":"GeoEdit: Geometry-Aware Object Editing via Dual-Branch Denoising","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30599","citing_title":"Goku: A Million-Scale Universal Dataset and Benchmark for Instruction-Based Video Editing","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00310","citing_title":"Where to Refine, When to Stop: Rethinking Redundancy via Latent Discrepancy for Efficient Visual Autoregressive Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22051","citing_title":"EasyVFX: Frequency-Driven Decoupling for Resource-Efficient VFX Generation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15803","citing_title":"Embedding-perturbed Exploration Preference Optimization for Flow Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16937","citing_title":"DEVIS-GRPO: Unleashing GRPO on Dynamic Extreme View Synthesis","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17019","citing_title":"StreamingEffect: Real-Time Human-Centric Video Effect Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08392","citing_title":"ST-BiBench: Benchmarking Multi-Stream Multimodal Coordination in Bimanual Embodied Tasks for MLLMs","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC","json":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC.json","graph_json":"https://pith.science/api/pith-number/YHRGUX6FP4U66NORN2UOVANQIC/graph.json","events_json":"https://pith.science/api/pith-number/YHRGUX6FP4U66NORN2UOVANQIC/events.json","paper":"https://pith.science/paper/YHRGUX6F"},"agent_actions":{"view_html":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC","download_json":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC.json","view_paper":"https://pith.science/paper/YHRGUX6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04590&json=true","fetch_graph":"https://pith.science/api/pith-number/YHRGUX6FP4U66NORN2UOVANQIC/graph.json","fetch_events":"https://pith.science/api/pith-number/YHRGUX6FP4U66NORN2UOVANQIC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC/action/storage_attestation","attest_author":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC/action/author_attestation","sign_citation":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC/action/citation_signature","submit_replication":"https://pith.science/pith/YHRGUX6FP4U66NORN2UOVANQIC/action/replication_record"}},"created_at":"2026-07-05T11:16:17.336429+00:00","updated_at":"2026-07-05T11:16:17.336429+00:00"}