{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RBEYH4CVKTRMFDCKBVLNTQKUZD","short_pith_number":"pith:RBEYH4CV","schema_version":"1.0","canonical_sha256":"884983f05554e2c28c4a0d56d9c154c8d65f03cd1f91a6bea213664d71f4c002","source":{"kind":"arxiv","id":"2411.17440","version":3},"attestation_state":"computed","paper":{"title":"Identity-Preserving Text-to-Video Generation by Frequency Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Jiebo Luo, Jinfa Huang, Liuhan Chen, Li Yuan, Shenghai Yuan, Xianyi He, Yujun Shi, Yunyuan Ge","submitted_at":"2024-11-26T13:58:24Z","abstract_excerpt":"Identity-preserving text-to-video (IPT2V) generation aims to create high-fidelity videos with consistent human identity. It is an important task in video generation but remains an open problem for generative models. This paper pushes the technical frontier of IPT2V in two directions that have not been resolved in literature: (1) A tuning-free pipeline without tedious case-by-case finetuning, and (2) A frequency-aware heuristic identity-preserving DiT-based control scheme. We propose ConsisID, a tuning-free DiT-based controllable IPT2V model to keep human identity consistent in the generated vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17440","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-26T13:58:24Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"1c5a7a136943f2a5b8385c367a6c9cb16fca5b0898539f59a225a4b5027e5ed9","abstract_canon_sha256":"89ce9f457cf2c91b8d6306af07f142ba53e20da6c05abb441fb311b56a936d9d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:23.335647Z","signature_b64":"1Y1GOc9K8FORpr0Is1UfosMGUJKoe1ITsVnOkbYlvS5EBdUpehvNh9hnoYCbMEEPwURXpDY79Ae9uphBOS6DBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"884983f05554e2c28c4a0d56d9c154c8d65f03cd1f91a6bea213664d71f4c002","last_reissued_at":"2026-07-05T10:39:23.335146Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:23.335146Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Identity-Preserving Text-to-Video Generation by Frequency Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Jiebo Luo, Jinfa Huang, Liuhan Chen, Li Yuan, Shenghai Yuan, Xianyi He, Yujun Shi, Yunyuan Ge","submitted_at":"2024-11-26T13:58:24Z","abstract_excerpt":"Identity-preserving text-to-video (IPT2V) generation aims to create high-fidelity videos with consistent human identity. It is an important task in video generation but remains an open problem for generative models. This paper pushes the technical frontier of IPT2V in two directions that have not been resolved in literature: (1) A tuning-free pipeline without tedious case-by-case finetuning, and (2) A frequency-aware heuristic identity-preserving DiT-based control scheme. We propose ConsisID, a tuning-free DiT-based controllable IPT2V model to keep human identity consistent in the generated vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17440","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17440/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17440","created_at":"2026-07-05T10:39:23.335204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17440v3","created_at":"2026-07-05T10:39:23.335204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17440","created_at":"2026-07-05T10:39:23.335204+00:00"},{"alias_kind":"pith_short_12","alias_value":"RBEYH4CVKTRM","created_at":"2026-07-05T10:39:23.335204+00:00"},{"alias_kind":"pith_short_16","alias_value":"RBEYH4CVKTRMFDCK","created_at":"2026-07-05T10:39:23.335204+00:00"},{"alias_kind":"pith_short_8","alias_value":"RBEYH4CV","created_at":"2026-07-05T10:39:23.335204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22347","citing_title":"Customizing Video Portraits via Identity-ActionDecoupling","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11783","citing_title":"A Comprehensive Ecosystem for Open-Domain Customized Video Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06903","citing_title":"Beyond Skeletons: Learning Animation Directly from Driving Videos with Same2X Training Strategy","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2504.17816","citing_title":"Learning Zero-Shot Subject-Driven Video Generation Using 1% Compute","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15684","citing_title":"ElasticDiT: Efficient Diffusion Transformers via Elastic Architecture and Sparse Attention for High-Resolution Image Generation on Mobile Devices","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23690","citing_title":"SynMotion: Semantic-Visual Adaptation for Motion Customized Video Generation","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13857","citing_title":"MoZoo:Unleashing Video Diffusion power in animal fur and muscle simulation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20275","citing_title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03147","citing_title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11244","citing_title":"Script-a-Video: Deep Structured Audio-visual Captions via Factorized Streams and Relational Grounding","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":196,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD","json":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD.json","graph_json":"https://pith.science/api/pith-number/RBEYH4CVKTRMFDCKBVLNTQKUZD/graph.json","events_json":"https://pith.science/api/pith-number/RBEYH4CVKTRMFDCKBVLNTQKUZD/events.json","paper":"https://pith.science/paper/RBEYH4CV"},"agent_actions":{"view_html":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD","download_json":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD.json","view_paper":"https://pith.science/paper/RBEYH4CV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17440&json=true","fetch_graph":"https://pith.science/api/pith-number/RBEYH4CVKTRMFDCKBVLNTQKUZD/graph.json","fetch_events":"https://pith.science/api/pith-number/RBEYH4CVKTRMFDCKBVLNTQKUZD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD/action/storage_attestation","attest_author":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD/action/author_attestation","sign_citation":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD/action/citation_signature","submit_replication":"https://pith.science/pith/RBEYH4CVKTRMFDCKBVLNTQKUZD/action/replication_record"}},"created_at":"2026-07-05T10:39:23.335204+00:00","updated_at":"2026-07-05T10:39:23.335204+00:00"}