{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZVA5OFX54OMR5CHIYHDYLSIYQV","short_pith_number":"pith:ZVA5OFX5","schema_version":"1.0","canonical_sha256":"cd41d716fde3991e88e8c1c785c91885408ce60c21acb2a543e3fbd7d2f99cd1","source":{"kind":"arxiv","id":"2412.14531","version":1},"attestation_state":"computed","paper":{"title":"Consistent Human Image and Video Generation with Spatially Conditioned Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Mou, Mingdeng Cao, Xintao Wang, Ying Shan, Yinqiang Zheng, Zhaoyang Zhang, Ziyang Yuan","submitted_at":"2024-12-19T05:02:30Z","abstract_excerpt":"Consistent human-centric image and video synthesis aims to generate images or videos with new poses while preserving appearance consistency with a given reference image, which is crucial for low-cost visual content creation. Recent advances based on diffusion models typically rely on separate networks for reference appearance feature extraction and target visual generation, leading to inconsistent domain gaps between references and targets. In this paper, we frame the task as a spatially-conditioned inpainting problem, where the target image is inpainted to maintain appearance consistency with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.14531","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-19T05:02:30Z","cross_cats_sorted":[],"title_canon_sha256":"53b28f19ac88d3ab6a3bff9a30aa7deadb16f84cefb91c88e857b774a8462cb6","abstract_canon_sha256":"638bd3b798987cc2a65b9aeeca469f89586efbefa91cedb0409935d9b341d3ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:48.944072Z","signature_b64":"iZ5RyWwYWWImMuQNACH4E4u1Y7dIbYua9HdJkpbyENTlsgR4I2+aOU69LbPj2mnmfPFVzWAahmQmBJewH+epCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd41d716fde3991e88e8c1c785c91885408ce60c21acb2a543e3fbd7d2f99cd1","last_reissued_at":"2026-07-05T09:51:48.943588Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:48.943588Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Consistent Human Image and Video Generation with Spatially Conditioned Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Mou, Mingdeng Cao, Xintao Wang, Ying Shan, Yinqiang Zheng, Zhaoyang Zhang, Ziyang Yuan","submitted_at":"2024-12-19T05:02:30Z","abstract_excerpt":"Consistent human-centric image and video synthesis aims to generate images or videos with new poses while preserving appearance consistency with a given reference image, which is crucial for low-cost visual content creation. Recent advances based on diffusion models typically rely on separate networks for reference appearance feature extraction and target visual generation, leading to inconsistent domain gaps between references and targets. In this paper, we frame the task as a spatially-conditioned inpainting problem, where the target image is inpainted to maintain appearance consistency with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.14531","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.14531/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.14531","created_at":"2026-07-05T09:51:48.943652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.14531v1","created_at":"2026-07-05T09:51:48.943652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.14531","created_at":"2026-07-05T09:51:48.943652+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZVA5OFX54OMR","created_at":"2026-07-05T09:51:48.943652+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZVA5OFX54OMR5CHI","created_at":"2026-07-05T09:51:48.943652+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZVA5OFX5","created_at":"2026-07-05T09:51:48.943652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20255","citing_title":"AniCrafter: Customizing Realistic Human-Centric Animation via Avatar-Background Conditioning in Video Diffusion Models","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV","json":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV.json","graph_json":"https://pith.science/api/pith-number/ZVA5OFX54OMR5CHIYHDYLSIYQV/graph.json","events_json":"https://pith.science/api/pith-number/ZVA5OFX54OMR5CHIYHDYLSIYQV/events.json","paper":"https://pith.science/paper/ZVA5OFX5"},"agent_actions":{"view_html":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV","download_json":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV.json","view_paper":"https://pith.science/paper/ZVA5OFX5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.14531&json=true","fetch_graph":"https://pith.science/api/pith-number/ZVA5OFX54OMR5CHIYHDYLSIYQV/graph.json","fetch_events":"https://pith.science/api/pith-number/ZVA5OFX54OMR5CHIYHDYLSIYQV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV/action/storage_attestation","attest_author":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV/action/author_attestation","sign_citation":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV/action/citation_signature","submit_replication":"https://pith.science/pith/ZVA5OFX54OMR5CHIYHDYLSIYQV/action/replication_record"}},"created_at":"2026-07-05T09:51:48.943652+00:00","updated_at":"2026-07-05T09:51:48.943652+00:00"}