{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6N5GSPBGRS34TEYG4HLSTPA52M","short_pith_number":"pith:6N5GSPBG","schema_version":"1.0","canonical_sha256":"f37a693c268cb7c99306e1d729bc1dd311dd67177bc8fcfbe37bcd5b9ac46f1c","source":{"kind":"arxiv","id":"2412.06974","version":1},"attestation_state":"computed","paper":{"title":"MV-DUSt3R+: Single-Stage Scene Reconstruction from Sparse Views In 2 Seconds","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Schwing, Dilin Wang, Hongyu Xu, Rakesh Ranjan, Yuchen Fan, Zhenggang Tang, Zhicheng Yan","submitted_at":"2024-12-09T20:34:55Z","abstract_excerpt":"Recent sparse multi-view scene reconstruction advances like DUSt3R and MASt3R no longer require camera calibration and camera pose estimation. However, they only process a pair of views at a time to infer pixel-aligned pointmaps. When dealing with more than two views, a combinatorial number of error prone pairwise reconstructions are usually followed by an expensive global optimization, which often fails to rectify the pairwise reconstruction errors. To handle more views, reduce errors, and improve inference time, we propose the fast single-stage feed-forward network MV-DUSt3R. At its core are"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06974","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-09T20:34:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3af72f3468755e834f6c436a4dbfe2e22d8836ea9fe6e86b4313d04c629d1262","abstract_canon_sha256":"54d481942b135c3f05b17b4ebe64d333296b35bec0202d78d2303209a305648e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:01.184722Z","signature_b64":"9PpOnw4cZvb5fnQenDJrc2UL14ut/xpxIU6V5/gg7IJSekTjqITJUHGBSDeR2cqHXgjt7JHCmmceefPzbxepAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f37a693c268cb7c99306e1d729bc1dd311dd67177bc8fcfbe37bcd5b9ac46f1c","last_reissued_at":"2026-07-05T09:47:01.184275Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:01.184275Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MV-DUSt3R+: Single-Stage Scene Reconstruction from Sparse Views In 2 Seconds","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Schwing, Dilin Wang, Hongyu Xu, Rakesh Ranjan, Yuchen Fan, Zhenggang Tang, Zhicheng Yan","submitted_at":"2024-12-09T20:34:55Z","abstract_excerpt":"Recent sparse multi-view scene reconstruction advances like DUSt3R and MASt3R no longer require camera calibration and camera pose estimation. However, they only process a pair of views at a time to infer pixel-aligned pointmaps. When dealing with more than two views, a combinatorial number of error prone pairwise reconstructions are usually followed by an expensive global optimization, which often fails to rectify the pairwise reconstruction errors. To handle more views, reduce errors, and improve inference time, we propose the fast single-stage feed-forward network MV-DUSt3R. At its core are"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06974","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06974","created_at":"2026-07-05T09:47:01.184337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06974v1","created_at":"2026-07-05T09:47:01.184337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06974","created_at":"2026-07-05T09:47:01.184337+00:00"},{"alias_kind":"pith_short_12","alias_value":"6N5GSPBGRS34","created_at":"2026-07-05T09:47:01.184337+00:00"},{"alias_kind":"pith_short_16","alias_value":"6N5GSPBGRS34TEYG","created_at":"2026-07-05T09:47:01.184337+00:00"},{"alias_kind":"pith_short_8","alias_value":"6N5GSPBG","created_at":"2026-07-05T09:47:01.184337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31124","citing_title":"QVGGT: Post-Training Quantized Visual Geometry Grounded Transformer","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31086","citing_title":"CasaMaestro: Multi-View Panoramas for House-Scale 3D Reconstruction","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26115","citing_title":"TriSplat: Simulation-Ready Feed-Forward 3D Scene Reconstruction","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12190","citing_title":"Bundle Adjustment in the Eager Mode","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20279","citing_title":"VLM-3R: Vision-Language Models Augmented with Instruction-Aligned 3D Reconstruction","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17568","citing_title":"PAGE-4D: Disentangled pose and geometry estimation for vggt-4d perception","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05959","citing_title":"OVGGT: O(1) Constant-Cost Streaming Visual Geometry Transformer","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2507.13347","citing_title":"$\\pi^3$: Permutation-Equivariant Visual Geometry Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08542","citing_title":"Scal3R: Scalable Test-Time Training for Large-Scale 3D Reconstruction","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08532","citing_title":"Self-Improving 4D Perception via Self-Distillation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14025","citing_title":"Feed-Forward 3D Scene Modeling: A Problem-Driven Perspective","ref_index":100,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M","json":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M.json","graph_json":"https://pith.science/api/pith-number/6N5GSPBGRS34TEYG4HLSTPA52M/graph.json","events_json":"https://pith.science/api/pith-number/6N5GSPBGRS34TEYG4HLSTPA52M/events.json","paper":"https://pith.science/paper/6N5GSPBG"},"agent_actions":{"view_html":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M","download_json":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M.json","view_paper":"https://pith.science/paper/6N5GSPBG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06974&json=true","fetch_graph":"https://pith.science/api/pith-number/6N5GSPBGRS34TEYG4HLSTPA52M/graph.json","fetch_events":"https://pith.science/api/pith-number/6N5GSPBGRS34TEYG4HLSTPA52M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M/action/storage_attestation","attest_author":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M/action/author_attestation","sign_citation":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M/action/citation_signature","submit_replication":"https://pith.science/pith/6N5GSPBGRS34TEYG4HLSTPA52M/action/replication_record"}},"created_at":"2026-07-05T09:47:01.184337+00:00","updated_at":"2026-07-05T09:47:01.184337+00:00"}