{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NPWWG44KQ5TME43J4JASO2RKTK","short_pith_number":"pith:NPWWG44K","schema_version":"1.0","canonical_sha256":"6bed63738a8766c27369e241276a2a9ab33814a1f277887894632c33db53eaee","source":{"kind":"arxiv","id":"2502.06145","version":1},"attestation_state":"computed","paper":{"title":"Animate Anyone 2: High-Fidelity Character Image Animation with Environment Affordance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bang Zhang, Dechao Meng, Guangyuan Wang, Lian Zhuo, Liefeng Bo, Li Hu, Peng Zhang, Xin Gao, Zhen Shen","submitted_at":"2025-02-10T04:20:11Z","abstract_excerpt":"Recent character image animation methods based on diffusion models, such as Animate Anyone, have made significant progress in generating consistent and generalizable character animations. However, these approaches fail to produce reasonable associations between characters and their environments. To address this limitation, we introduce Animate Anyone 2, aiming to animate characters with environment affordance. Beyond extracting motion signals from source video, we additionally capture environmental representations as conditional inputs. The environment is formulated as the region with the excl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06145","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-10T04:20:11Z","cross_cats_sorted":[],"title_canon_sha256":"23b5936d68b88fe146691a8240defdd2ceb1dc4b7a6d7468a7cfe4eca599729b","abstract_canon_sha256":"ac02a3378e8df9ecdeb0da8a8be18501aab4a070d875367e5d9120d831f6a605"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:48.337221Z","signature_b64":"6xkev33rikQA0dw92/47KXFat854/wP3Avkxhe3Xc2vOxuhhJ4hCerpjdQdYsnc9asJBw+Oq1n/4wjLKdjr9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6bed63738a8766c27369e241276a2a9ab33814a1f277887894632c33db53eaee","last_reissued_at":"2026-07-05T10:11:48.336687Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:48.336687Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Animate Anyone 2: High-Fidelity Character Image Animation with Environment Affordance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bang Zhang, Dechao Meng, Guangyuan Wang, Lian Zhuo, Liefeng Bo, Li Hu, Peng Zhang, Xin Gao, Zhen Shen","submitted_at":"2025-02-10T04:20:11Z","abstract_excerpt":"Recent character image animation methods based on diffusion models, such as Animate Anyone, have made significant progress in generating consistent and generalizable character animations. However, these approaches fail to produce reasonable associations between characters and their environments. To address this limitation, we introduce Animate Anyone 2, aiming to animate characters with environment affordance. Beyond extracting motion signals from source video, we additionally capture environmental representations as conditional inputs. The environment is formulated as the region with the excl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06145","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06145","created_at":"2026-07-05T10:11:48.336758+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06145v1","created_at":"2026-07-05T10:11:48.336758+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06145","created_at":"2026-07-05T10:11:48.336758+00:00"},{"alias_kind":"pith_short_12","alias_value":"NPWWG44KQ5TM","created_at":"2026-07-05T10:11:48.336758+00:00"},{"alias_kind":"pith_short_16","alias_value":"NPWWG44KQ5TME43J","created_at":"2026-07-05T10:11:48.336758+00:00"},{"alias_kind":"pith_short_8","alias_value":"NPWWG44K","created_at":"2026-07-05T10:11:48.336758+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30514","citing_title":"3D Scene-Adaptive Trajectory-Controllable Human Image Animation with Camera Movement","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02000","citing_title":"Towards 3D-Aware Video Diffusion Models: Render-Free Human Motion Control with Mesh Tokenization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30514","citing_title":"3D Scene-Adaptive Trajectory-Controllable Human Image Animation with Camera Movement","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19320","citing_title":"SteadyDancer: Harmonized and Coherent Human Image Animation with First-Frame Preservation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17248","citing_title":"Image-to-Video Diffusion: From Foundations to Open Frontiers","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07951","citing_title":"Preserving Source Video Realism: High-Fidelity Face Swapping for Cinematic Quality","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2512.09646","citing_title":"VHOI: Controllable Video Generation of Human-Object Interactions from Sparse Trajectories via Motion Densification","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22160","citing_title":"Screen, Cache, and Match: A Training-Free Causality-Consistent Reference Frame Framework for Human Animation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":273,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05961","citing_title":"HumANDiff: Articulated Noise Diffusion for Motion-Consistent Human Video Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21291","citing_title":"Exploring the Role of Synthetic Data Augmentation in Controllable Human-Centric Video Generation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK","json":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK.json","graph_json":"https://pith.science/api/pith-number/NPWWG44KQ5TME43J4JASO2RKTK/graph.json","events_json":"https://pith.science/api/pith-number/NPWWG44KQ5TME43J4JASO2RKTK/events.json","paper":"https://pith.science/paper/NPWWG44K"},"agent_actions":{"view_html":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK","download_json":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK.json","view_paper":"https://pith.science/paper/NPWWG44K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06145&json=true","fetch_graph":"https://pith.science/api/pith-number/NPWWG44KQ5TME43J4JASO2RKTK/graph.json","fetch_events":"https://pith.science/api/pith-number/NPWWG44KQ5TME43J4JASO2RKTK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK/action/storage_attestation","attest_author":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK/action/author_attestation","sign_citation":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK/action/citation_signature","submit_replication":"https://pith.science/pith/NPWWG44KQ5TME43J4JASO2RKTK/action/replication_record"}},"created_at":"2026-07-05T10:11:48.336758+00:00","updated_at":"2026-07-05T10:11:48.336758+00:00"}