{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VP5LMXDXTANNIDWQYIICSMNA7T","short_pith_number":"pith:VP5LMXDX","schema_version":"1.0","canonical_sha256":"abfab65c77981ad40ed0c2102931a0fcfe74924433477801de81633d43148683","source":{"kind":"arxiv","id":"2412.19645","version":2},"attestation_state":"computed","paper":{"title":"VideoMaker: Zero-shot Customized Video Generation with the Inherent Force of Video Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangcong Zheng, Huanzhang Dou, Junfu Pu, Tao Wu, Xiaodong Cun, Xi Li, Ying Shan, Yong Zhang, Zhongang Qi","submitted_at":"2024-12-27T13:49:25Z","abstract_excerpt":"Zero-shot customized video generation has gained significant attention due to its substantial application potential. Existing methods rely on additional models to extract and inject reference subject features, assuming that the Video Diffusion Model (VDM) alone is insufficient for zero-shot customized video generation. However, these methods often struggle to maintain consistent subject appearance due to suboptimal feature extraction and injection techniques. In this paper, we reveal that VDM inherently possesses the force to extract and inject subject features. Departing from previous heurist"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.19645","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-27T13:49:25Z","cross_cats_sorted":[],"title_canon_sha256":"5a47591b5e7e6c037265cde09d89e7e4fe4af8ec9de8ef2fbddeeb113390efd0","abstract_canon_sha256":"f0cddc2d34a981f70ca8a41242d10a334351f444afb97127f5ab72f2ef8fc3e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:16.592814Z","signature_b64":"loaGJt9TKnVaNRULwDVNv/52607U098HGqifkELWXqieVnHnHts0ftkW7kMzoV28njOWS18HdSkQVPuGbnwTDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abfab65c77981ad40ed0c2102931a0fcfe74924433477801de81633d43148683","last_reissued_at":"2026-07-05T09:55:16.592348Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:16.592348Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoMaker: Zero-shot Customized Video Generation with the Inherent Force of Video Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangcong Zheng, Huanzhang Dou, Junfu Pu, Tao Wu, Xiaodong Cun, Xi Li, Ying Shan, Yong Zhang, Zhongang Qi","submitted_at":"2024-12-27T13:49:25Z","abstract_excerpt":"Zero-shot customized video generation has gained significant attention due to its substantial application potential. Existing methods rely on additional models to extract and inject reference subject features, assuming that the Video Diffusion Model (VDM) alone is insufficient for zero-shot customized video generation. However, these methods often struggle to maintain consistent subject appearance due to suboptimal feature extraction and injection techniques. In this paper, we reveal that VDM inherently possesses the force to extract and inject subject features. Departing from previous heurist"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.19645","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.19645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.19645","created_at":"2026-07-05T09:55:16.592408+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.19645v2","created_at":"2026-07-05T09:55:16.592408+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.19645","created_at":"2026-07-05T09:55:16.592408+00:00"},{"alias_kind":"pith_short_12","alias_value":"VP5LMXDXTANN","created_at":"2026-07-05T09:55:16.592408+00:00"},{"alias_kind":"pith_short_16","alias_value":"VP5LMXDXTANNIDWQ","created_at":"2026-07-05T09:55:16.592408+00:00"},{"alias_kind":"pith_short_8","alias_value":"VP5LMXDX","created_at":"2026-07-05T09:55:16.592408+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11670","citing_title":"ARGUS: Stacked Multi-View Identity Mosaic Injection for Subject-Preserving Video Generation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17248","citing_title":"Image-to-Video Diffusion: From Foundations to Open Frontiers","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T","json":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T.json","graph_json":"https://pith.science/api/pith-number/VP5LMXDXTANNIDWQYIICSMNA7T/graph.json","events_json":"https://pith.science/api/pith-number/VP5LMXDXTANNIDWQYIICSMNA7T/events.json","paper":"https://pith.science/paper/VP5LMXDX"},"agent_actions":{"view_html":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T","download_json":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T.json","view_paper":"https://pith.science/paper/VP5LMXDX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.19645&json=true","fetch_graph":"https://pith.science/api/pith-number/VP5LMXDXTANNIDWQYIICSMNA7T/graph.json","fetch_events":"https://pith.science/api/pith-number/VP5LMXDXTANNIDWQYIICSMNA7T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T/action/storage_attestation","attest_author":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T/action/author_attestation","sign_citation":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T/action/citation_signature","submit_replication":"https://pith.science/pith/VP5LMXDXTANNIDWQYIICSMNA7T/action/replication_record"}},"created_at":"2026-07-05T09:55:16.592408+00:00","updated_at":"2026-07-05T09:55:16.592408+00:00"}