{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7I2YIZINAYNAVERJSR4KWQ744G","short_pith_number":"pith:7I2YIZIN","schema_version":"1.0","canonical_sha256":"fa3584650d061a0a92299478ab43fce193bb0a043f622e52628e72744ab368b7","source":{"kind":"arxiv","id":"2401.02473","version":1},"attestation_state":"computed","paper":{"title":"VASE: Object-Centric Appearance and Shape Manipulation of Real Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dejia Xu, Elia Peruzzo, Humphrey Shi, Nicu Sebe, Vidit Goel, Xingqian Xu, Yifan Jiang, Zhangyang Wang","submitted_at":"2024-01-04T18:59:24Z","abstract_excerpt":"Recently, several works tackled the video editing task fostered by the success of large-scale text-to-image generative models. However, most of these methods holistically edit the frame using the text, exploiting the prior given by foundation diffusion models and focusing on improving the temporal consistency across frames. In this work, we introduce a framework that is object-centric and is designed to control both the object's appearance and, notably, to execute precise and explicit structural modifications on the object. We build our framework on a pre-trained image-conditioned diffusion mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.02473","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-04T18:59:24Z","cross_cats_sorted":[],"title_canon_sha256":"a2d63a7571631ed96e495c64c1b0324e1071f378d8ba5fc5d8955958f56cbba0","abstract_canon_sha256":"6554728024888535d24aa6168fe832d8cc0adeff51c8a40b213af5f68ce2cddc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:30:25.911802Z","signature_b64":"YeY1uY2+LViIx3IqWnhRhpcszt9nLj5BCbKC+0GI85ja2S+HZgdflBZO5hfEiFyaZPps4URou1XOE9BhenNTDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa3584650d061a0a92299478ab43fce193bb0a043f622e52628e72744ab368b7","last_reissued_at":"2026-07-05T07:30:25.911251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:30:25.911251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VASE: Object-Centric Appearance and Shape Manipulation of Real Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dejia Xu, Elia Peruzzo, Humphrey Shi, Nicu Sebe, Vidit Goel, Xingqian Xu, Yifan Jiang, Zhangyang Wang","submitted_at":"2024-01-04T18:59:24Z","abstract_excerpt":"Recently, several works tackled the video editing task fostered by the success of large-scale text-to-image generative models. However, most of these methods holistically edit the frame using the text, exploiting the prior given by foundation diffusion models and focusing on improving the temporal consistency across frames. In this work, we introduce a framework that is object-centric and is designed to control both the object's appearance and, notably, to execute precise and explicit structural modifications on the object. We build our framework on a pre-trained image-conditioned diffusion mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.02473","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02473/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.02473","created_at":"2026-07-05T07:30:25.911319+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.02473v1","created_at":"2026-07-05T07:30:25.911319+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02473","created_at":"2026-07-05T07:30:25.911319+00:00"},{"alias_kind":"pith_short_12","alias_value":"7I2YIZINAYNA","created_at":"2026-07-05T07:30:25.911319+00:00"},{"alias_kind":"pith_short_16","alias_value":"7I2YIZINAYNAVERJ","created_at":"2026-07-05T07:30:25.911319+00:00"},{"alias_kind":"pith_short_8","alias_value":"7I2YIZIN","created_at":"2026-07-05T07:30:25.911319+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.00503","citing_title":"Diff4Splat: Controllable 4D Scene Generation with Latent Dynamic Reconstruction Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02509","citing_title":"CamCo: Camera-Controllable 3D-Consistent Image-to-Video Generation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02048","citing_title":"ViewCrafter: Taming Video Diffusion Models for High-fidelity Novel View Synthesis","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":272,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G","json":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G.json","graph_json":"https://pith.science/api/pith-number/7I2YIZINAYNAVERJSR4KWQ744G/graph.json","events_json":"https://pith.science/api/pith-number/7I2YIZINAYNAVERJSR4KWQ744G/events.json","paper":"https://pith.science/paper/7I2YIZIN"},"agent_actions":{"view_html":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G","download_json":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G.json","view_paper":"https://pith.science/paper/7I2YIZIN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.02473&json=true","fetch_graph":"https://pith.science/api/pith-number/7I2YIZINAYNAVERJSR4KWQ744G/graph.json","fetch_events":"https://pith.science/api/pith-number/7I2YIZINAYNAVERJSR4KWQ744G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G/action/storage_attestation","attest_author":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G/action/author_attestation","sign_citation":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G/action/citation_signature","submit_replication":"https://pith.science/pith/7I2YIZINAYNAVERJSR4KWQ744G/action/replication_record"}},"created_at":"2026-07-05T07:30:25.911319+00:00","updated_at":"2026-07-05T07:30:25.911319+00:00"}