{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M57NZSDAKB6NICBYGYPMPPJ66S","short_pith_number":"pith:M57NZSDA","schema_version":"1.0","canonical_sha256":"677edcc860507cd40838361ec7bd3ef49e63b48ead5c7118e4b18c93ce3385be","source":{"kind":"arxiv","id":"2502.13995","version":1},"attestation_state":"computed","paper":{"title":"FantasyID: Face Knowledge Enhanced ID-Preserving Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.GR","authors_text":"Fan Jiang, Mu Xu, Qiang Wang, Yaqi Fan, Yonggang Qi, Yunpeng Zhang","submitted_at":"2025-02-19T06:50:27Z","abstract_excerpt":"Tuning-free approaches adapting large-scale pre-trained video diffusion models for identity-preserving text-to-video generation (IPT2V) have gained popularity recently due to their efficacy and scalability. However, significant challenges remain to achieve satisfied facial dynamics while keeping the identity unchanged. In this work, we present a novel tuning-free IPT2V framework by enhancing face knowledge of the pre-trained video model built on diffusion transformers (DiT), dubbed FantasyID. Essentially, 3D facial geometry prior is incorporated to ensure plausible facial structures during vid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.13995","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.GR","submitted_at":"2025-02-19T06:50:27Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"18b72d4b7b06fdf719b9296bbe72e8fd3c27d0e8dd3408108851b821c6584bcc","abstract_canon_sha256":"e53a8668be19ba8515403e08a866322ee44d0e9132a1bb3ff5a9b0f2d0de76f2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:39.610653Z","signature_b64":"Omhg0ajYif29etPJBg+3mKVgTI9y4jHBBlZoGA5o7FvRjNFFcuJ+25WudILPujRYRTapyKTeKcb/dsby1lTPAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"677edcc860507cd40838361ec7bd3ef49e63b48ead5c7118e4b18c93ce3385be","last_reissued_at":"2026-07-05T10:19:39.610170Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:39.610170Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FantasyID: Face Knowledge Enhanced ID-Preserving Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.GR","authors_text":"Fan Jiang, Mu Xu, Qiang Wang, Yaqi Fan, Yonggang Qi, Yunpeng Zhang","submitted_at":"2025-02-19T06:50:27Z","abstract_excerpt":"Tuning-free approaches adapting large-scale pre-trained video diffusion models for identity-preserving text-to-video generation (IPT2V) have gained popularity recently due to their efficacy and scalability. However, significant challenges remain to achieve satisfied facial dynamics while keeping the identity unchanged. In this work, we present a novel tuning-free IPT2V framework by enhancing face knowledge of the pre-trained video model built on diffusion transformers (DiT), dubbed FantasyID. Essentially, 3D facial geometry prior is incorporated to ensure plausible facial structures during vid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13995","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13995/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.13995","created_at":"2026-07-05T10:19:39.610228+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.13995v1","created_at":"2026-07-05T10:19:39.610228+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13995","created_at":"2026-07-05T10:19:39.610228+00:00"},{"alias_kind":"pith_short_12","alias_value":"M57NZSDAKB6N","created_at":"2026-07-05T10:19:39.610228+00:00"},{"alias_kind":"pith_short_16","alias_value":"M57NZSDAKB6NICBY","created_at":"2026-07-05T10:19:39.610228+00:00"},{"alias_kind":"pith_short_8","alias_value":"M57NZSDA","created_at":"2026-07-05T10:19:39.610228+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11670","citing_title":"ARGUS: Stacked Multi-View Identity Mosaic Injection for Subject-Preserving Video Generation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04702","citing_title":"FaithfulFaces: Pose-Faithful Facial Identity Preservation for Text-to-Video Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":198,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S","json":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S.json","graph_json":"https://pith.science/api/pith-number/M57NZSDAKB6NICBYGYPMPPJ66S/graph.json","events_json":"https://pith.science/api/pith-number/M57NZSDAKB6NICBYGYPMPPJ66S/events.json","paper":"https://pith.science/paper/M57NZSDA"},"agent_actions":{"view_html":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S","download_json":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S.json","view_paper":"https://pith.science/paper/M57NZSDA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.13995&json=true","fetch_graph":"https://pith.science/api/pith-number/M57NZSDAKB6NICBYGYPMPPJ66S/graph.json","fetch_events":"https://pith.science/api/pith-number/M57NZSDAKB6NICBYGYPMPPJ66S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S/action/storage_attestation","attest_author":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S/action/author_attestation","sign_citation":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S/action/citation_signature","submit_replication":"https://pith.science/pith/M57NZSDAKB6NICBYGYPMPPJ66S/action/replication_record"}},"created_at":"2026-07-05T10:19:39.610228+00:00","updated_at":"2026-07-05T10:19:39.610228+00:00"}