{"work":{"id":"7c214f4a-d3ff-48e7-bd13-043eb9c139cb","openalex_id":null,"doi":null,"arxiv_id":"2507.07982","raw_key":null,"title":"Geometry Forcing: Marrying Video Diffusion and 3D Representation for Consistent World Modeling","authors":null,"authors_text":null,"year":2025,"venue":"cs.CV","abstract":"Videos inherently represent 2D projections of a dynamic 3D world. However, our analysis suggests that video diffusion models trained solely on raw video data often fail to capture meaningful geometric-aware structure in their learned representations. To bridge the gap between video diffusion models and the underlying 3D nature of the physical world, we propose Geometry Forcing, a simple yet effective method that encourages video diffusion models to internalize 3D representations. Our key insight is to guide the model's intermediate representations toward geometry-aware structure by aligning them with features from a geometric foundation model. To this end, we introduce two complementary alignment objectives: Angular Alignment, which enforces directional consistency via cosine similarity, and Scale Alignment, which preserves scale-related information by regressing geometric features from normalized diffusion representations. We evaluate Geometry Forcing on both camera-view conditioned and action-conditioned video generation tasks. Experimental results demonstrate that our method substantially improves visual quality and 3D consistency over the baseline methods. Project page: https://GeometryForcing.github.io.","external_url":"https://arxiv.org/abs/2507.07982","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-07T14:33:54.305490+00:00","pith_arxiv_id":"2507.07982","created_at":"2026-05-10T10:09:08.139387+00:00","updated_at":"2026-07-07T14:33:54.305490+00:00","title_quality_ok":true,"display_title":"Geometry Forcing: Marrying Video Diffusion and 3D Representation for Consistent World Modeling","render_title":"Geometry Forcing: Marrying Video Diffusion and 3D Representation for Consistent World Modeling"},"hub":{"state":{"work_id":"7c214f4a-d3ff-48e7-bd13-043eb9c139cb","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":27,"external_cited_by_count":null,"distinct_field_count":1,"first_pith_cited_at":"2026-04-05T10:47:52+00:00","last_pith_cited_at":"2026-07-06T17:51:00+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-19T13:19:51.269210+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":8},{"context_role":"baseline","n":1},{"context_role":"method","n":1}],"polarity_counts":[{"context_polarity":"background","n":8},{"context_polarity":"baseline","n":1},{"context_polarity":"use_method","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}