{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:C47QNCKDISXDRLVJWZHDK5GOOY","short_pith_number":"pith:C47QNCKD","schema_version":"1.0","canonical_sha256":"173f06894344ae38aea9b64e3574ce7600cc63e12e93bc6b0374cfe1552a729b","source":{"kind":"arxiv","id":"2501.12375","version":3},"attestation_state":"computed","paper":{"title":"Video Depth Anything: Consistent Depth Estimation for Super-Long Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Feihu Zhang, Hengkai Guo, Jiashi Feng, Shengnan Zhu, Sili Chen, Zilong Huang","submitted_at":"2025-01-21T18:53:30Z","abstract_excerpt":"Depth Anything has achieved remarkable success in monocular depth estimation with strong generalization ability. However, it suffers from temporal inconsistency in videos, hindering its practical applications. Various methods have been proposed to alleviate this issue by leveraging video generation models or introducing priors from optical flow and camera poses. Nonetheless, these methods are only applicable to short videos (< 10 seconds) and require a trade-off between quality and computational efficiency. We propose Video Depth Anything for high-quality, consistent depth estimation in super-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.12375","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-21T18:53:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7ea7457f6415fe92547c9b9535fc0b40d7ed621188089ac773878cbab3b6582b","abstract_canon_sha256":"c802c5cc5b0b151753e7c1df0f248383ede0fa068dddbe5605a24b2b679bc830"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:40.813797Z","signature_b64":"hwy/ZiNd2dxmeeqp7VwRzhXJbkNYeKz7oes57yxNHXGa6JGGQUfSYj1FozLb5EwzU5cT/2kNhp2m87vci276Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"173f06894344ae38aea9b64e3574ce7600cc63e12e93bc6b0374cfe1552a729b","last_reissued_at":"2026-07-05T11:21:40.813241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:40.813241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Depth Anything: Consistent Depth Estimation for Super-Long Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Feihu Zhang, Hengkai Guo, Jiashi Feng, Shengnan Zhu, Sili Chen, Zilong Huang","submitted_at":"2025-01-21T18:53:30Z","abstract_excerpt":"Depth Anything has achieved remarkable success in monocular depth estimation with strong generalization ability. However, it suffers from temporal inconsistency in videos, hindering its practical applications. Various methods have been proposed to alleviate this issue by leveraging video generation models or introducing priors from optical flow and camera poses. Nonetheless, these methods are only applicable to short videos (< 10 seconds) and require a trade-off between quality and computational efficiency. We propose Video Depth Anything for high-quality, consistent depth estimation in super-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.12375","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.12375/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.12375","created_at":"2026-07-05T11:21:40.813302+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.12375v3","created_at":"2026-07-05T11:21:40.813302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.12375","created_at":"2026-07-05T11:21:40.813302+00:00"},{"alias_kind":"pith_short_12","alias_value":"C47QNCKDISXD","created_at":"2026-07-05T11:21:40.813302+00:00"},{"alias_kind":"pith_short_16","alias_value":"C47QNCKDISXDRLVJ","created_at":"2026-07-05T11:21:40.813302+00:00"},{"alias_kind":"pith_short_8","alias_value":"C47QNCKD","created_at":"2026-07-05T11:21:40.813302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22160","citing_title":"GenMatter: Perceiving Physical Objects with Generative Matter Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26410","citing_title":"Neural Voxel Dynamics: Learning Implicit 3D Physics via Volumetric Feature Advection","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25308","citing_title":"Stabilizing Streaming Video Geometry via Dynamic Feature Normalization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28128","citing_title":"PhysisForcing: Physics Reinforced World Simulator for Robotic Manipulation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23098","citing_title":"UfM*: Uncertainty from Motion* for DNN Depth Estimation Using Gaussians","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2411.14295","citing_title":"DissolveStereo: Coarse Depth Injection for Zero-Shot Stereo Video Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2512.11988","citing_title":"CARI4D: Category Agnostic 4D Reconstruction of Human-Object Interaction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10934","citing_title":"ViPE: Video Pose Engine for 3D Geometric Perception","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04016","citing_title":"HOIGS: Human-Object Interaction Gaussian Splatting","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24718","citing_title":"WildLIFT: Lifting monocular drone video to 3D for species-agnostic wildlife monitoring","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22160","citing_title":"GenMatter: Perceiving Physical Objects with Generative Matter Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00658","citing_title":"UniVidX: A Unified Multimodal Framework for Versatile Video Generation via Diffusion Priors","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09352","citing_title":"LuMon: A Comprehensive Benchmark and Development Suite with Novel Datasets for Lunar Monocular Depth Estimation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14556","citing_title":"Controllable Video Object Insertion via Multi-View Priors","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY","json":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY.json","graph_json":"https://pith.science/api/pith-number/C47QNCKDISXDRLVJWZHDK5GOOY/graph.json","events_json":"https://pith.science/api/pith-number/C47QNCKDISXDRLVJWZHDK5GOOY/events.json","paper":"https://pith.science/paper/C47QNCKD"},"agent_actions":{"view_html":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY","download_json":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY.json","view_paper":"https://pith.science/paper/C47QNCKD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.12375&json=true","fetch_graph":"https://pith.science/api/pith-number/C47QNCKDISXDRLVJWZHDK5GOOY/graph.json","fetch_events":"https://pith.science/api/pith-number/C47QNCKDISXDRLVJWZHDK5GOOY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY/action/storage_attestation","attest_author":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY/action/author_attestation","sign_citation":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY/action/citation_signature","submit_replication":"https://pith.science/pith/C47QNCKDISXDRLVJWZHDK5GOOY/action/replication_record"}},"created_at":"2026-07-05T11:21:40.813302+00:00","updated_at":"2026-07-05T11:21:40.813302+00:00"}