{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OVJQSIK2APFGLX3X5DJPKO45TD","short_pith_number":"pith:OVJQSIK2","schema_version":"1.0","canonical_sha256":"755309215a03ca65df77e8d2f53b9d98ebc4ec6bc09c8050f1e4f3dab2d106f0","source":{"kind":"arxiv","id":"2307.14460","version":1},"attestation_state":"computed","paper":{"title":"MiDaS v3.1 -- A Model Zoo for Robust Monocular Relative Depth Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Diana Wofk, Matthias M\\\"uller, Reiner Birkl","submitted_at":"2023-07-26T19:01:49Z","abstract_excerpt":"We release MiDaS v3.1 for monocular depth estimation, offering a variety of new models based on different encoder backbones. This release is motivated by the success of transformers in computer vision, with a large variety of pretrained vision transformers now available. We explore how using the most promising vision transformers as image encoders impacts depth estimation quality and runtime of the MiDaS architecture. Our investigation also includes recent convolutional approaches that achieve comparable quality to vision transformers in image classification tasks. While the previous release M"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.14460","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-07-26T19:01:49Z","cross_cats_sorted":[],"title_canon_sha256":"fa7dab42d91db6a2d9e23a869218d1e4abce849f42bfa40c578da1ed6ecaed6a","abstract_canon_sha256":"a4a350df2488edeeaf9956202df11e4d07c6d5078a608692017c8095ec598080"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:13.505852Z","signature_b64":"iCZX2leYWgwUt+/oP9Q8Q07n+L7BYKb/yLMI7DlZ00C1fOxVA3V9QTSBBP0Rc/XWEaEIgU77QFSBlfkc3j3jCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"755309215a03ca65df77e8d2f53b9d98ebc4ec6bc09c8050f1e4f3dab2d106f0","last_reissued_at":"2026-07-05T06:35:13.505420Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:13.505420Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MiDaS v3.1 -- A Model Zoo for Robust Monocular Relative Depth Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Diana Wofk, Matthias M\\\"uller, Reiner Birkl","submitted_at":"2023-07-26T19:01:49Z","abstract_excerpt":"We release MiDaS v3.1 for monocular depth estimation, offering a variety of new models based on different encoder backbones. This release is motivated by the success of transformers in computer vision, with a large variety of pretrained vision transformers now available. We explore how using the most promising vision transformers as image encoders impacts depth estimation quality and runtime of the MiDaS architecture. Our investigation also includes recent convolutional approaches that achieve comparable quality to vision transformers in image classification tasks. While the previous release M"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.14460","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.14460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.14460","created_at":"2026-07-05T06:35:13.505486+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.14460v1","created_at":"2026-07-05T06:35:13.505486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.14460","created_at":"2026-07-05T06:35:13.505486+00:00"},{"alias_kind":"pith_short_12","alias_value":"OVJQSIK2APFG","created_at":"2026-07-05T06:35:13.505486+00:00"},{"alias_kind":"pith_short_16","alias_value":"OVJQSIK2APFGLX3X","created_at":"2026-07-05T06:35:13.505486+00:00"},{"alias_kind":"pith_short_8","alias_value":"OVJQSIK2","created_at":"2026-07-05T06:35:13.505486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06779","citing_title":"URS-Stereo: Uncertainty-Guided Residual Search for Real-Time Stereo Matching","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25503","citing_title":"AISPO: Enhancing Depth Reliability for Robotic Manipulation of Non-Lambertian Objects via Affine-Invariant Shape Prior","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21300","citing_title":"SCOPE: Scale-Consistent One-Pass Estimation of 3D Geometry","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07813","citing_title":"MinNav: Minimalist Navigation Using Optical Flow For Active Tiny Aerial Robots","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29600","citing_title":"One Scene, Two Depths: Probing Geometric Ambiguity in Monocular Foundation Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29716","citing_title":"AerialMetric: Benchmarking and Adapting UAV Monocular Metric Depth Estimation in the Real World","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25308","citing_title":"Stabilizing Streaming Video Geometry via Dynamic Feature Normalization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26636","citing_title":"JetViT: Efficient High-Resolution Vision Transformer with Post-Training Attention Search","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30060","citing_title":"Towards Consistent Video Geometry Estimation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2412.00131","citing_title":"Open-Sora Plan: Open-Source Large Video Generation Model","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2501.03717","citing_title":"Materialist: Physically Based Editing Using Single-Image Inverse Rendering","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10525","citing_title":"GemDepth: Geometry-Embedded Features for 3D-Consistent Video Depth","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19539","citing_title":"Trust It or Not: Evidential Uncertainty for Feed-Forward 3D Reconstruction with Trust3R","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17893","citing_title":"LUMEN: Low-light Unified Multi-stage Enhancement Network using depth-guided flash, clustering, and attention-based Transformers","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2506.20616","citing_title":"Shape2Animal: Creative Animal Generation from Natural Silhouettes","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":199,"is_internal_anchor":false},{"citing_arxiv_id":"2406.16860","citing_title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01765","citing_title":"Efficient Test-Time Optimization for Depth Completion via Low-Rank Decoder Adaptation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22686","citing_title":"SS3D: End2End Self-Supervised 3D from Web Videos","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10525","citing_title":"GemDepth: Geometry-Embedded Features for 3D-Consistent Video Depth","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02546","citing_title":"MoGe-2: Accurate Monocular Geometry with Metric Scale and Sharp Details","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2406.09414","citing_title":"Depth Anything V2","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10525","citing_title":"GemDepth: Geometry-Embedded Features for 3D-Consistent Video Depth","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12494","citing_title":"Revisiting Photometric Ambiguity for Accurate Gaussian-Splatting Surface Reconstruction","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11578","citing_title":"The Midas Touch for Metric Depth","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD","json":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD.json","graph_json":"https://pith.science/api/pith-number/OVJQSIK2APFGLX3X5DJPKO45TD/graph.json","events_json":"https://pith.science/api/pith-number/OVJQSIK2APFGLX3X5DJPKO45TD/events.json","paper":"https://pith.science/paper/OVJQSIK2"},"agent_actions":{"view_html":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD","download_json":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD.json","view_paper":"https://pith.science/paper/OVJQSIK2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.14460&json=true","fetch_graph":"https://pith.science/api/pith-number/OVJQSIK2APFGLX3X5DJPKO45TD/graph.json","fetch_events":"https://pith.science/api/pith-number/OVJQSIK2APFGLX3X5DJPKO45TD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD/action/storage_attestation","attest_author":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD/action/author_attestation","sign_citation":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD/action/citation_signature","submit_replication":"https://pith.science/pith/OVJQSIK2APFGLX3X5DJPKO45TD/action/replication_record"}},"created_at":"2026-07-05T06:35:13.505486+00:00","updated_at":"2026-07-05T06:35:13.505486+00:00"}