{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OUK4W4KCEN65Z333YHPUH7E5FL","short_pith_number":"pith:OUK4W4KC","schema_version":"1.0","canonical_sha256":"7515cb7142237ddcef7bc1df43fc9d2afa762b9ae4745256eeeba6c5e7ed7db8","source":{"kind":"arxiv","id":"2410.10815","version":2},"attestation_state":"computed","paper":{"title":"Depth Any Video with Scalable Synthetic Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Binbin Lin, Chunhua Shen, Di Huang, Haifeng Liu, Honghui Yang, Tong He, Wanli Ouyang, Wei Yin, Xiaofei He","submitted_at":"2024-10-14T17:59:46Z","abstract_excerpt":"Video depth estimation has long been hindered by the scarcity of consistent and scalable ground truth data, leading to inconsistent and unreliable results. In this paper, we introduce Depth Any Video, a model that tackles the challenge through two key innovations. First, we develop a scalable synthetic data pipeline, capturing real-time video depth data from diverse virtual environments, yielding 40,000 video clips of 5-second duration, each with precise depth annotations. Second, we leverage the powerful priors of generative video diffusion models to handle real-world videos effectively, inte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.10815","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-14T17:59:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"82694c507940e952727051f6e63e2c78d39baaa167b02a9bd3982f0c00c6776e","abstract_canon_sha256":"6284eb496d7ee87d6e80e8d3551aad20230263ee87b6418fc1a873bb3ce6b2e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:29:24.827369Z","signature_b64":"oUYVWsXBk7bf2FgIiXVR7tvrsGFab2uFVyiyBIvcbY+5fogfcqgFvtSWvI3lq1ob/opFfXE6p0gn58RA6Bk0BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7515cb7142237ddcef7bc1df43fc9d2afa762b9ae4745256eeeba6c5e7ed7db8","last_reissued_at":"2026-07-05T10:29:24.826723Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:29:24.826723Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Depth Any Video with Scalable Synthetic Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Binbin Lin, Chunhua Shen, Di Huang, Haifeng Liu, Honghui Yang, Tong He, Wanli Ouyang, Wei Yin, Xiaofei He","submitted_at":"2024-10-14T17:59:46Z","abstract_excerpt":"Video depth estimation has long been hindered by the scarcity of consistent and scalable ground truth data, leading to inconsistent and unreliable results. In this paper, we introduce Depth Any Video, a model that tackles the challenge through two key innovations. First, we develop a scalable synthetic data pipeline, capturing real-time video depth data from diverse virtual environments, yielding 40,000 video clips of 5-second duration, each with precise depth annotations. Second, we leverage the powerful priors of generative video diffusion models to handle real-world videos effectively, inte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.10815","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.10815/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.10815","created_at":"2026-07-05T10:29:24.826808+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.10815v2","created_at":"2026-07-05T10:29:24.826808+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.10815","created_at":"2026-07-05T10:29:24.826808+00:00"},{"alias_kind":"pith_short_12","alias_value":"OUK4W4KCEN65","created_at":"2026-07-05T10:29:24.826808+00:00"},{"alias_kind":"pith_short_16","alias_value":"OUK4W4KCEN65Z333","created_at":"2026-07-05T10:29:24.826808+00:00"},{"alias_kind":"pith_short_8","alias_value":"OUK4W4KC","created_at":"2026-07-05T10:29:24.826808+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30060","citing_title":"Towards Consistent Video Geometry Estimation","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20306","citing_title":"TPGDiff: Hierarchical Triple-Prior Guided Diffusion for Image Restoration","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17844","citing_title":"Less is More: Data-Efficient Adaptation for Controllable Text-to-Video Generation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00658","citing_title":"UniVidX: A Unified Multimodal Framework for Versatile Video Generation via Diffusion Priors","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06665","citing_title":"VDPP: Video Depth Post-Processing for Speed and Scalability","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL","json":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL.json","graph_json":"https://pith.science/api/pith-number/OUK4W4KCEN65Z333YHPUH7E5FL/graph.json","events_json":"https://pith.science/api/pith-number/OUK4W4KCEN65Z333YHPUH7E5FL/events.json","paper":"https://pith.science/paper/OUK4W4KC"},"agent_actions":{"view_html":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL","download_json":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL.json","view_paper":"https://pith.science/paper/OUK4W4KC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.10815&json=true","fetch_graph":"https://pith.science/api/pith-number/OUK4W4KCEN65Z333YHPUH7E5FL/graph.json","fetch_events":"https://pith.science/api/pith-number/OUK4W4KCEN65Z333YHPUH7E5FL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL/action/storage_attestation","attest_author":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL/action/author_attestation","sign_citation":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL/action/citation_signature","submit_replication":"https://pith.science/pith/OUK4W4KCEN65Z333YHPUH7E5FL/action/replication_record"}},"created_at":"2026-07-05T10:29:24.826808+00:00","updated_at":"2026-07-05T10:29:24.826808+00:00"}