{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WGPYUEVNWQ5LIGCWDP2S53GAZL","short_pith_number":"pith:WGPYUEVN","schema_version":"1.0","canonical_sha256":"b19f8a12adb43ab418561bf52eecc0cad2e785acab61f41ea30eeb539c39c310","source":{"kind":"arxiv","id":"2412.10533","version":1},"attestation_state":"computed","paper":{"title":"SUGAR: Subject-Driven Video Customization in a Zero-Shot Manner","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jing Shi, Jiuxiang Gu, Nanxuan Zhao, Ruiyi Zhang, Tong Sun, Yufan Zhou","submitted_at":"2024-12-13T20:01:51Z","abstract_excerpt":"We present SUGAR, a zero-shot method for subject-driven video customization. Given an input image, SUGAR is capable of generating videos for the subject contained in the image and aligning the generation with arbitrary visual attributes such as style and motion specified by user-input text. Unlike previous methods, which require test-time fine-tuning or fail to generate text-aligned videos, SUGAR achieves superior results without the need for extra cost at test-time. To enable zero-shot capability, we introduce a scalable pipeline to construct synthetic dataset which is specifically designed f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.10533","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-13T20:01:51Z","cross_cats_sorted":[],"title_canon_sha256":"b05d6b785e734f58162da615ba581df07862c70386c2812096464e69309a7094","abstract_canon_sha256":"c993ce4e610b2685707700ee9bb9aeab84ff0247b6abf9f2418f8af601686983"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:17.316068Z","signature_b64":"zXdOG3kn9z7Vy5FWFY6MCC9jRUHt8vxfEpAMJNPAA09pn27+Eo5Qwf5jhBCNQSVY0bcXcM3DmsyafLwTIFQ5DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b19f8a12adb43ab418561bf52eecc0cad2e785acab61f41ea30eeb539c39c310","last_reissued_at":"2026-07-05T09:49:17.315602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:17.315602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SUGAR: Subject-Driven Video Customization in a Zero-Shot Manner","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jing Shi, Jiuxiang Gu, Nanxuan Zhao, Ruiyi Zhang, Tong Sun, Yufan Zhou","submitted_at":"2024-12-13T20:01:51Z","abstract_excerpt":"We present SUGAR, a zero-shot method for subject-driven video customization. Given an input image, SUGAR is capable of generating videos for the subject contained in the image and aligning the generation with arbitrary visual attributes such as style and motion specified by user-input text. Unlike previous methods, which require test-time fine-tuning or fail to generate text-aligned videos, SUGAR achieves superior results without the need for extra cost at test-time. To enable zero-shot capability, we introduce a scalable pipeline to construct synthetic dataset which is specifically designed f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.10533","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.10533/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.10533","created_at":"2026-07-05T09:49:17.315656+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.10533v1","created_at":"2026-07-05T09:49:17.315656+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.10533","created_at":"2026-07-05T09:49:17.315656+00:00"},{"alias_kind":"pith_short_12","alias_value":"WGPYUEVNWQ5L","created_at":"2026-07-05T09:49:17.315656+00:00"},{"alias_kind":"pith_short_16","alias_value":"WGPYUEVNWQ5LIGCW","created_at":"2026-07-05T09:49:17.315656+00:00"},{"alias_kind":"pith_short_8","alias_value":"WGPYUEVN","created_at":"2026-07-05T09:49:17.315656+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":110,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL","json":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL.json","graph_json":"https://pith.science/api/pith-number/WGPYUEVNWQ5LIGCWDP2S53GAZL/graph.json","events_json":"https://pith.science/api/pith-number/WGPYUEVNWQ5LIGCWDP2S53GAZL/events.json","paper":"https://pith.science/paper/WGPYUEVN"},"agent_actions":{"view_html":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL","download_json":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL.json","view_paper":"https://pith.science/paper/WGPYUEVN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.10533&json=true","fetch_graph":"https://pith.science/api/pith-number/WGPYUEVNWQ5LIGCWDP2S53GAZL/graph.json","fetch_events":"https://pith.science/api/pith-number/WGPYUEVNWQ5LIGCWDP2S53GAZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL/action/storage_attestation","attest_author":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL/action/author_attestation","sign_citation":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL/action/citation_signature","submit_replication":"https://pith.science/pith/WGPYUEVNWQ5LIGCWDP2S53GAZL/action/replication_record"}},"created_at":"2026-07-05T09:49:17.315656+00:00","updated_at":"2026-07-05T09:49:17.315656+00:00"}