{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:S4F62QH3W2ZHEMM5S7U5ZLB7C6","short_pith_number":"pith:S4F62QH3","schema_version":"1.0","canonical_sha256":"970bed40fbb6b272319d97e9dcac3f17a4e828fda4ca5c28270cb9916e9328de","source":{"kind":"arxiv","id":"2503.03751","version":1},"attestation_state":"computed","paper":{"title":"GEN3C: 3D-Informed World-Consistent Video Generation with Precise Camera Control","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Alexander Keller, Huan Ling, Jiahui Huang, Jun Gao, Merlin Nimier-David, Sanja Fidler, Thomas M\\\"uller, Tianchang Shen, Xuanchi Ren, Yifan Lu","submitted_at":"2025-03-05T18:59:50Z","abstract_excerpt":"We present GEN3C, a generative video model with precise Camera Control and temporal 3D Consistency. Prior video models already generate realistic videos, but they tend to leverage little 3D information, leading to inconsistencies, such as objects popping in and out of existence. Camera control, if implemented at all, is imprecise, because camera parameters are mere inputs to the neural network which must then infer how the video depends on the camera. In contrast, GEN3C is guided by a 3D cache: point clouds obtained by predicting the pixel-wise depth of seed images or previously generated fram"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.03751","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-05T18:59:50Z","cross_cats_sorted":["cs.GR"],"title_canon_sha256":"50375d2ae3eb8b81416faacedcd14f7bb8069762dbdb2bd3b0af0f7624380450","abstract_canon_sha256":"a5bef2ac4f993fed1e16469ef3d22227585e7f6d574dd85439949a0859c0f131"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:00.081862Z","signature_b64":"9g1NOzaSeZW+DYu/+9lhDFYXcV87lBHrBKU6ApZgBub62RWJRmIfDqKJf013X1b92yRjS6etZSAxcLZXD9z6Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"970bed40fbb6b272319d97e9dcac3f17a4e828fda4ca5c28270cb9916e9328de","last_reissued_at":"2026-07-05T10:25:00.081198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:00.081198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GEN3C: 3D-Informed World-Consistent Video Generation with Precise Camera Control","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GR"],"primary_cat":"cs.CV","authors_text":"Alexander Keller, Huan Ling, Jiahui Huang, Jun Gao, Merlin Nimier-David, Sanja Fidler, Thomas M\\\"uller, Tianchang Shen, Xuanchi Ren, Yifan Lu","submitted_at":"2025-03-05T18:59:50Z","abstract_excerpt":"We present GEN3C, a generative video model with precise Camera Control and temporal 3D Consistency. Prior video models already generate realistic videos, but they tend to leverage little 3D information, leading to inconsistencies, such as objects popping in and out of existence. Camera control, if implemented at all, is imprecise, because camera parameters are mere inputs to the neural network which must then infer how the video depends on the camera. In contrast, GEN3C is guided by a 3D cache: point clouds obtained by predicting the pixel-wise depth of seed images or previously generated fram"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.03751","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.03751/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.03751","created_at":"2026-07-05T10:25:00.081267+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.03751v1","created_at":"2026-07-05T10:25:00.081267+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.03751","created_at":"2026-07-05T10:25:00.081267+00:00"},{"alias_kind":"pith_short_12","alias_value":"S4F62QH3W2ZH","created_at":"2026-07-05T10:25:00.081267+00:00"},{"alias_kind":"pith_short_16","alias_value":"S4F62QH3W2ZHEMM5","created_at":"2026-07-05T10:25:00.081267+00:00"},{"alias_kind":"pith_short_8","alias_value":"S4F62QH3","created_at":"2026-07-05T10:25:00.081267+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31734","citing_title":"MemLearner: Learning to Query Context memory for Video World Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22481","citing_title":"Lighting-Consistent Object Transfer Across Radiance Fields","ref_index":250,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21996","citing_title":"VRAG: Learning World Models for Interactive Video Generation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02467","citing_title":"VERTIGO: Visual Preference Optimization for Cinematic Camera Trajectory Generation","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6","json":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6.json","graph_json":"https://pith.science/api/pith-number/S4F62QH3W2ZHEMM5S7U5ZLB7C6/graph.json","events_json":"https://pith.science/api/pith-number/S4F62QH3W2ZHEMM5S7U5ZLB7C6/events.json","paper":"https://pith.science/paper/S4F62QH3"},"agent_actions":{"view_html":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6","download_json":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6.json","view_paper":"https://pith.science/paper/S4F62QH3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.03751&json=true","fetch_graph":"https://pith.science/api/pith-number/S4F62QH3W2ZHEMM5S7U5ZLB7C6/graph.json","fetch_events":"https://pith.science/api/pith-number/S4F62QH3W2ZHEMM5S7U5ZLB7C6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6/action/storage_attestation","attest_author":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6/action/author_attestation","sign_citation":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6/action/citation_signature","submit_replication":"https://pith.science/pith/S4F62QH3W2ZHEMM5S7U5ZLB7C6/action/replication_record"}},"created_at":"2026-07-05T10:25:00.081267+00:00","updated_at":"2026-07-05T10:25:00.081267+00:00"}