{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y7VMBZVT5K2N4COWE4GVFUG3QP","short_pith_number":"pith:Y7VMBZVT","schema_version":"1.0","canonical_sha256":"c7eac0e6b3eab4de09d6270d52d0db83e383a379dc82e421e59a0df3405e66ed","source":{"kind":"arxiv","id":"2409.02095","version":2},"attestation_state":"computed","paper":{"title":"DepthCrafter: Generating Consistent Long Depth Sequences for Open-world Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.GR"],"primary_cat":"cs.CV","authors_text":"Long Quan, Sijie Zhao, Wenbo Hu, Xiangjun Gao, Xiaodong Cun, Xiaoyu Li, Ying Shan, Yong Zhang","submitted_at":"2024-09-03T17:52:03Z","abstract_excerpt":"Estimating video depth in open-world scenarios is challenging due to the diversity of videos in appearance, content motion, camera movement, and length. We present DepthCrafter, an innovative method for generating temporally consistent long depth sequences with intricate details for open-world videos, without requiring any supplementary information such as camera poses or optical flow. The generalization ability to open-world videos is achieved by training the video-to-depth model from a pre-trained image-to-video diffusion model, through our meticulously designed three-stage training strategy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.02095","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-03T17:52:03Z","cross_cats_sorted":["cs.AI","cs.GR"],"title_canon_sha256":"d8a746e9fd49cff4c32af420992f61f1ec6f97667db4885ba294818899591bfb","abstract_canon_sha256":"8511a7fc363349583604d36ffa10f21bd790c48542e464b7642af896c147f5c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:40.913916Z","signature_b64":"q2ayc3dAcjAUN+6KAAL0rD4dabQloVa1QcGrZNjD8ReJLaitMe9RxN4nlk26nUzbUcroiuv3Y5YNJGUspQsVAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7eac0e6b3eab4de09d6270d52d0db83e383a379dc82e421e59a0df3405e66ed","last_reissued_at":"2026-07-05T09:41:40.913436Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:40.913436Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DepthCrafter: Generating Consistent Long Depth Sequences for Open-world Videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.GR"],"primary_cat":"cs.CV","authors_text":"Long Quan, Sijie Zhao, Wenbo Hu, Xiangjun Gao, Xiaodong Cun, Xiaoyu Li, Ying Shan, Yong Zhang","submitted_at":"2024-09-03T17:52:03Z","abstract_excerpt":"Estimating video depth in open-world scenarios is challenging due to the diversity of videos in appearance, content motion, camera movement, and length. We present DepthCrafter, an innovative method for generating temporally consistent long depth sequences with intricate details for open-world videos, without requiring any supplementary information such as camera poses or optical flow. The generalization ability to open-world videos is achieved by training the video-to-depth model from a pre-trained image-to-video diffusion model, through our meticulously designed three-stage training strategy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.02095","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.02095/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.02095","created_at":"2026-07-05T09:41:40.913493+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.02095v2","created_at":"2026-07-05T09:41:40.913493+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.02095","created_at":"2026-07-05T09:41:40.913493+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7VMBZVT5K2N","created_at":"2026-07-05T09:41:40.913493+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7VMBZVT5K2N4COW","created_at":"2026-07-05T09:41:40.913493+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7VMBZVT","created_at":"2026-07-05T09:41:40.913493+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08771","citing_title":"ZipDepth: Bringing Lightweight Zero-Shot Monocular Depth Anywhere, on Any Device","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25308","citing_title":"Stabilizing Streaming Video Geometry via Dynamic Feature Normalization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30060","citing_title":"Towards Consistent Video Geometry Estimation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23888","citing_title":"GenRecon: Bridging Generative Priors for Multi-View 3D Scene Reconstruction","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2411.14295","citing_title":"DissolveStereo: Coarse Depth Injection for Zero-Shot Stereo Video Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17182","citing_title":"Radar-Guided Polynomial Fitting for Metric Depth Estimation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01099","citing_title":"Geometry-aware 4D Video Generation for Robot Manipulation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2410.03825","citing_title":"MonST3R: A Simple Approach for Estimating Geometry in the Presence of Motion","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12774","citing_title":"WildPose: A Unified Framework for Robust Pose Estimation in the Wild","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP","json":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP.json","graph_json":"https://pith.science/api/pith-number/Y7VMBZVT5K2N4COWE4GVFUG3QP/graph.json","events_json":"https://pith.science/api/pith-number/Y7VMBZVT5K2N4COWE4GVFUG3QP/events.json","paper":"https://pith.science/paper/Y7VMBZVT"},"agent_actions":{"view_html":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP","download_json":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP.json","view_paper":"https://pith.science/paper/Y7VMBZVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.02095&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7VMBZVT5K2N4COWE4GVFUG3QP/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7VMBZVT5K2N4COWE4GVFUG3QP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP/action/storage_attestation","attest_author":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP/action/author_attestation","sign_citation":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP/action/citation_signature","submit_replication":"https://pith.science/pith/Y7VMBZVT5K2N4COWE4GVFUG3QP/action/replication_record"}},"created_at":"2026-07-05T09:41:40.913493+00:00","updated_at":"2026-07-05T09:41:40.913493+00:00"}