{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OLFAK2RPMRT5ZUTUSAXF27O3WY","short_pith_number":"pith:OLFAK2RP","schema_version":"1.0","canonical_sha256":"72ca056a2f6467dcd274902e5d7ddbb60ed1ab3ee834b89dd98f68f3652b3f9c","source":{"kind":"arxiv","id":"2406.02511","version":1},"attestation_state":"computed","paper":{"title":"V-Express: Conditional Dropout for Progressive Training of Portrait Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Cong Wang, Fei Shen, Feng Luo, Jun Zhang, Kuan Tian, Qing Gu, Wei Yang, Xiao Han, Yonghang Guan, Zhiwei Jiang","submitted_at":"2024-06-04T17:32:52Z","abstract_excerpt":"In the field of portrait video generation, the use of single images to generate portrait videos has become increasingly prevalent. A common approach involves leveraging generative models to enhance adapters for controlled generation. However, control signals (e.g., text, audio, reference image, pose, depth map, etc.) can vary in strength. Among these, weaker conditions often struggle to be effective due to interference from stronger conditions, posing a challenge in balancing these conditions. In our work on portrait video generation, we identified audio signals as particularly weak, often ove"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02511","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-04T17:32:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"da7a200ebc54a06a45c90316b7f118ed8506f218015838ddf3a01e105e1e3268","abstract_canon_sha256":"2e102dab32c37bba68cc6bdce8d803b43e149fc92accb26449011707c273fee9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:20.798697Z","signature_b64":"8adK1O2lk5EJH3NK7mtLxExGO20EXlIkjtOjQxs2r4kXqKLc6WD1SVCFvMVGNvEbd7/0zwS3M6OZwQ+QDsjHBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72ca056a2f6467dcd274902e5d7ddbb60ed1ab3ee834b89dd98f68f3652b3f9c","last_reissued_at":"2026-07-05T08:27:20.798230Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:20.798230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"V-Express: Conditional Dropout for Progressive Training of Portrait Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Cong Wang, Fei Shen, Feng Luo, Jun Zhang, Kuan Tian, Qing Gu, Wei Yang, Xiao Han, Yonghang Guan, Zhiwei Jiang","submitted_at":"2024-06-04T17:32:52Z","abstract_excerpt":"In the field of portrait video generation, the use of single images to generate portrait videos has become increasingly prevalent. A common approach involves leveraging generative models to enhance adapters for controlled generation. However, control signals (e.g., text, audio, reference image, pose, depth map, etc.) can vary in strength. Among these, weaker conditions often struggle to be effective due to interference from stronger conditions, posing a challenge in balancing these conditions. In our work on portrait video generation, we identified audio signals as particularly weak, often ove"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02511","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02511/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02511","created_at":"2026-07-05T08:27:20.798282+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02511v1","created_at":"2026-07-05T08:27:20.798282+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02511","created_at":"2026-07-05T08:27:20.798282+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLFAK2RPMRT5","created_at":"2026-07-05T08:27:20.798282+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLFAK2RPMRT5ZUTU","created_at":"2026-07-05T08:27:20.798282+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLFAK2RP","created_at":"2026-07-05T08:27:20.798282+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01031","citing_title":"Temporally-Aligned Evaluation for Audio-Driven Talking Head Generation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24176","citing_title":"Loki: Representation over Architecture for Diffusion-Based Portrait Animation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17248","citing_title":"Image-to-Video Diffusion: From Foundations to Open Frontiers","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12052","citing_title":"FluentAvatar: Flicker-Free Talking-Head Animation via Phoneme-Guided Autoregressive Modeling","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05539","citing_title":"VDCook:DIY video data cook your MLLMs","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY","json":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY.json","graph_json":"https://pith.science/api/pith-number/OLFAK2RPMRT5ZUTUSAXF27O3WY/graph.json","events_json":"https://pith.science/api/pith-number/OLFAK2RPMRT5ZUTUSAXF27O3WY/events.json","paper":"https://pith.science/paper/OLFAK2RP"},"agent_actions":{"view_html":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY","download_json":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY.json","view_paper":"https://pith.science/paper/OLFAK2RP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02511&json=true","fetch_graph":"https://pith.science/api/pith-number/OLFAK2RPMRT5ZUTUSAXF27O3WY/graph.json","fetch_events":"https://pith.science/api/pith-number/OLFAK2RPMRT5ZUTUSAXF27O3WY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY/action/storage_attestation","attest_author":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY/action/author_attestation","sign_citation":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY/action/citation_signature","submit_replication":"https://pith.science/pith/OLFAK2RPMRT5ZUTUSAXF27O3WY/action/replication_record"}},"created_at":"2026-07-05T08:27:20.798282+00:00","updated_at":"2026-07-05T08:27:20.798282+00:00"}