{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:T4MDMYDQJZPW2P7PIAORTBJ5VO","short_pith_number":"pith:T4MDMYDQ","schema_version":"1.0","canonical_sha256":"9f183660704e5f6d3fef401d19853daba5e8554e9af009ab04b9d1b7847575f8","source":{"kind":"arxiv","id":"2508.20210","version":1},"attestation_state":"computed","paper":{"title":"InfinityHuman: Towards Long-Term Audio-Driven Human","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyue Peng, Chen Zhang, Fangyuan Kong, Pan Xie, Qijun Gan, Xiang Yin, Xiaodi Li, Yi Ren, Zehuan Yuan","submitted_at":"2025-08-27T18:36:30Z","abstract_excerpt":"Audio-driven human animation has attracted wide attention thanks to its practical applications. However, critical challenges remain in generating high-resolution, long-duration videos with consistent appearance and natural hand motions. Existing methods extend videos using overlapping motion frames but suffer from error accumulation, leading to identity drift, color shifts, and scene instability. Additionally, hand movements are poorly modeled, resulting in noticeable distortions and misalignment with the audio. In this work, we propose InfinityHuman, a coarse-to-fine framework that first gene"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.20210","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-27T18:36:30Z","cross_cats_sorted":[],"title_canon_sha256":"c038d3ab73ff00457347cac2d5d17f343f866aa5e4be50ea581d4d42a2e97d17","abstract_canon_sha256":"0f303ccada2b0386a83b5e0ae96d12b80e78c43163ba531f9d1ca46f2dd4e707"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:00:31.366461Z","signature_b64":"RiY2JGN+RwT4OexO4qAfRIP/S1q2YwhBJh6XoYNeYtAXUBvwkmf/fYfdrMAG51e/BAdSzi5gyHT4q/mbT7M2Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f183660704e5f6d3fef401d19853daba5e8554e9af009ab04b9d1b7847575f8","last_reissued_at":"2026-07-05T12:00:31.365966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:00:31.365966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InfinityHuman: Towards Long-Term Audio-Driven Human","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyue Peng, Chen Zhang, Fangyuan Kong, Pan Xie, Qijun Gan, Xiang Yin, Xiaodi Li, Yi Ren, Zehuan Yuan","submitted_at":"2025-08-27T18:36:30Z","abstract_excerpt":"Audio-driven human animation has attracted wide attention thanks to its practical applications. However, critical challenges remain in generating high-resolution, long-duration videos with consistent appearance and natural hand motions. Existing methods extend videos using overlapping motion frames but suffer from error accumulation, leading to identity drift, color shifts, and scene instability. Additionally, hand movements are poorly modeled, resulting in noticeable distortions and misalignment with the audio. In this work, we propose InfinityHuman, a coarse-to-fine framework that first gene"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.20210","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.20210/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.20210","created_at":"2026-07-05T12:00:31.366022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.20210v1","created_at":"2026-07-05T12:00:31.366022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.20210","created_at":"2026-07-05T12:00:31.366022+00:00"},{"alias_kind":"pith_short_12","alias_value":"T4MDMYDQJZPW","created_at":"2026-07-05T12:00:31.366022+00:00"},{"alias_kind":"pith_short_16","alias_value":"T4MDMYDQJZPW2P7P","created_at":"2026-07-05T12:00:31.366022+00:00"},{"alias_kind":"pith_short_8","alias_value":"T4MDMYDQ","created_at":"2026-07-05T12:00:31.366022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30849","citing_title":"SyncCache: Exploiting Asymmetric Dynamics for Fast Audio-Driven Portrait Animation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14234","citing_title":"ViBES: A Conversational Agent with Behaviorally-Intelligent 3D Virtual Body","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13669","citing_title":"EchoTorrent: Towards Swift, Sustained, and Streaming Multi-Modal Video Generation","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27918","citing_title":"Generate Your Talking Avatar from Video Reference","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO","json":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO.json","graph_json":"https://pith.science/api/pith-number/T4MDMYDQJZPW2P7PIAORTBJ5VO/graph.json","events_json":"https://pith.science/api/pith-number/T4MDMYDQJZPW2P7PIAORTBJ5VO/events.json","paper":"https://pith.science/paper/T4MDMYDQ"},"agent_actions":{"view_html":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO","download_json":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO.json","view_paper":"https://pith.science/paper/T4MDMYDQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.20210&json=true","fetch_graph":"https://pith.science/api/pith-number/T4MDMYDQJZPW2P7PIAORTBJ5VO/graph.json","fetch_events":"https://pith.science/api/pith-number/T4MDMYDQJZPW2P7PIAORTBJ5VO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO/action/storage_attestation","attest_author":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO/action/author_attestation","sign_citation":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO/action/citation_signature","submit_replication":"https://pith.science/pith/T4MDMYDQJZPW2P7PIAORTBJ5VO/action/replication_record"}},"created_at":"2026-07-05T12:00:31.366022+00:00","updated_at":"2026-07-05T12:00:31.366022+00:00"}