{"work":{"id":"dc22e9ff-fcf5-4069-8ace-35ae3a0bfd7c","openalex_id":"https://openalex.org/W4416551737","doi":"10.48550/arxiv.2511.16624","arxiv_id":"2511.16624","raw_key":null,"title":"SAM 3D: 3Dfy Anything in Images","authors":null,"authors_text":"SAM 3D Team, Xingyu Chen, Fu-Jen Chu, Pierre Gleize, Kevin J Liang, Alexander Sax","year":2025,"venue":"cs.CV","abstract":"We present SAM 3D, a generative model for visually grounded 3D object reconstruction, predicting geometry, texture, and layout from a single image. SAM 3D excels in natural images, where occlusion and scene clutter are common and visual recognition cues from context play a larger role. We achieve this with a human- and model-in-the-loop pipeline for annotating object shape, texture, and pose, providing visually grounded 3D reconstruction data at unprecedented scale. We learn from this data in a modern, multi-stage training framework that combines synthetic pretraining with real-world alignment, breaking the 3D \"data barrier\". We obtain significant gains over recent work, with at least a 5:1 win rate in human preference tests on real-world objects and scenes. We will release our code and model weights, an online demo, and a new challenging benchmark for in-the-wild 3D object reconstruction.","external_url":"https://arxiv.org/abs/2511.16624","cited_by_count":0,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2511.16624","created_at":"2026-05-10T00:29:47.477812+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":false,"display_title":"SAM 3D: 3Dfy Anything in Images","render_title":"SAM 3D: 3Dfy Anything in Images"},"hub":{"state":{"work_id":"dc22e9ff-fcf5-4069-8ace-35ae3a0bfd7c","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":101,"external_cited_by_count":0,"distinct_field_count":6,"first_pith_cited_at":"2026-01-16T09:11:55+00:00","last_pith_cited_at":"2026-07-08T16:20:19+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-23T04:19:27.467688+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":11},{"context_role":"method","n":6},{"context_role":"baseline","n":3}],"polarity_counts":[{"context_polarity":"background","n":11},{"context_polarity":"use_method","n":6},{"context_polarity":"baseline","n":3}],"runs":{"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-14T18:00:19.800725+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"InstantMesh: Efficient 3D Mesh Generation from a Single Image with Sparse-view Large Reconstruction Models","work_id":"fc55eabb-0871-4dc4-8bab-b572e0d2aac4","shared_citers":8},{"title":"SAM 3: Segment Anything with Concepts","work_id":"4a72a006-2592-4554-aad0-a9c41a9f952d","shared_citers":8},{"title":"Depth Anything 3: Recovering the Visual Space from Any Views","work_id":"0a54b500-1e9d-46c2-85eb-8e16cbac8461","shared_citers":6},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":6},{"title":"DreamFusion: Text-to-3D using 2D Diffusion","work_id":"7529df29-9980-4a8a-b55e-307e3c2f357b","shared_citers":6},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":6},{"title":"arXiv2506.15442(2025) 10","work_id":"ee52f4d7-462f-4491-9549-4160820ae563","shared_citers":5},{"title":"arXiv preprint arXiv:2412.01506 (2024) 4","work_id":"6ce98e46-048f-47fa-917a-0c001eaae143","shared_citers":5},{"title":"DINOv3","work_id":"c8b07deb-8fe7-4e18-9620-f3569d3529ce","shared_citers":5},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":5},{"title":"Wan: Open and Advanced Large-Scale Video Generative Models","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","shared_citers":5},{"title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","shared_citers":4},{"title":"arXiv2502.06608(2025) 5, 6, 10","work_id":"bf744acd-a6e4-4ba7-98eb-43811d95fa81","shared_citers":4},{"title":"arXiv preprint arXiv:2212.08751(2022)","work_id":"9d7f0b29-b9ca-457f-9518-4a1506b43369","shared_citers":4},{"title":"arXiv preprint arXiv:2506.16504 (2025)","work_id":"94fe94cc-10de-4093-be65-3170f7e638cb","shared_citers":4},{"title":"arXiv preprint arXiv:2508.15769 (2025)","work_id":"83d3fcf2-2eae-4ce7-b7d9-9911bcb76d2b","shared_citers":4},{"title":"CoRR , volume =","work_id":"78943aa6-b9e9-4fc1-8d71-e6bcc0b9a0b4","shared_citers":4},{"title":"Hi3dgen: High-fidelity 3d geometry generation from im- ages via normal bridging.arXiv preprint arXiv:2503.22236","work_id":"67aa4434-ef33-4b27-ae4f-1e5fa4225745","shared_citers":4},{"title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","shared_citers":4},{"title":"Lrm: Large reconstruction model for single image to 3d","work_id":"0662dc2c-cc1c-4358-99bd-2a5f34795738","shared_citers":4},{"title":"OpenVLA: An Open-Source Vision-Language-Action Model","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","shared_citers":4},{"title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","work_id":"d1ad7304-d09a-49bc-809e-846439f6aff9","shared_citers":3},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":3},{"title":"arXiv preprint arXiv:2111.08897 (2021) 4, 8","work_id":"0ce910be-ca1c-44c7-b7b1-c5353759d85e","shared_citers":3}],"time_series":[{"n":36,"year":2026}],"dependency_candidates":[]},"error":null,"updated_at":"2026-05-14T18:00:04.631441+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"items":[{"title":"Qwen3 Technical Report","outcome":"unchanged","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"counts":{"fixed":0,"merged":0,"unchanged":1,"quarantined":0,"needs_external_resolution":0},"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-14T18:00:28.525876+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"SAM 3D: 3Dfy Anything in Images","claims":[{"claim_text":"We present SAM 3D, a generative model for visually grounded 3D object reconstruction, predicting geometry, texture, and layout from a single image. SAM 3D excels in natural images, where occlusion and scene clutter are common and visual recognition cues from context play a larger role. We achieve this with a human- and model-in-the-loop pipeline for annotating object shape, texture, and pose, providing visually grounded 3D reconstruction data at unprecedented scale. We learn from this data in a modern, multi-stage training framework that combines synthetic pretraining with real-world alignment","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks SAM 3D: 3Dfy Anything in Images because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T18:00:28.554845+00:00"}},"summary":{"title":"SAM 3D: 3Dfy Anything in Images","claims":[{"claim_text":"We present SAM 3D, a generative model for visually grounded 3D object reconstruction, predicting geometry, texture, and layout from a single image. SAM 3D excels in natural images, where occlusion and scene clutter are common and visual recognition cues from context play a larger role. We achieve this with a human- and model-in-the-loop pipeline for annotating object shape, texture, and pose, providing visually grounded 3D reconstruction data at unprecedented scale. We learn from this data in a modern, multi-stage training framework that combines synthetic pretraining with real-world alignment","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks SAM 3D: 3Dfy Anything in Images because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"InstantMesh: Efficient 3D Mesh Generation from a Single Image with Sparse-view Large Reconstruction Models","work_id":"fc55eabb-0871-4dc4-8bab-b572e0d2aac4","shared_citers":8},{"title":"SAM 3: Segment Anything with Concepts","work_id":"4a72a006-2592-4554-aad0-a9c41a9f952d","shared_citers":8},{"title":"Depth Anything 3: Recovering the Visual Space from Any Views","work_id":"0a54b500-1e9d-46c2-85eb-8e16cbac8461","shared_citers":6},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":6},{"title":"DreamFusion: Text-to-3D using 2D Diffusion","work_id":"7529df29-9980-4a8a-b55e-307e3c2f357b","shared_citers":6},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":6},{"title":"arXiv2506.15442(2025) 10","work_id":"ee52f4d7-462f-4491-9549-4160820ae563","shared_citers":5},{"title":"arXiv preprint arXiv:2412.01506 (2024) 4","work_id":"6ce98e46-048f-47fa-917a-0c001eaae143","shared_citers":5},{"title":"DINOv3","work_id":"c8b07deb-8fe7-4e18-9620-f3569d3529ce","shared_citers":5},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":5},{"title":"Wan: Open and Advanced Large-Scale Video Generative Models","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","shared_citers":5},{"title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","work_id":"f790abdc-a796-482f-a40d-f8ee035ecfc2","shared_citers":4},{"title":"arXiv2502.06608(2025) 5, 6, 10","work_id":"bf744acd-a6e4-4ba7-98eb-43811d95fa81","shared_citers":4},{"title":"arXiv preprint arXiv:2212.08751(2022)","work_id":"9d7f0b29-b9ca-457f-9518-4a1506b43369","shared_citers":4},{"title":"arXiv preprint arXiv:2506.16504 (2025)","work_id":"94fe94cc-10de-4093-be65-3170f7e638cb","shared_citers":4},{"title":"arXiv preprint arXiv:2508.15769 (2025)","work_id":"83d3fcf2-2eae-4ce7-b7d9-9911bcb76d2b","shared_citers":4},{"title":"CoRR , volume =","work_id":"78943aa6-b9e9-4fc1-8d71-e6bcc0b9a0b4","shared_citers":4},{"title":"Hi3dgen: High-fidelity 3d geometry generation from im- ages via normal bridging.arXiv preprint arXiv:2503.22236","work_id":"67aa4434-ef33-4b27-ae4f-1e5fa4225745","shared_citers":4},{"title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","shared_citers":4},{"title":"Lrm: Large reconstruction model for single image to 3d","work_id":"0662dc2c-cc1c-4358-99bd-2a5f34795738","shared_citers":4},{"title":"OpenVLA: An Open-Source Vision-Language-Action Model","work_id":"3e7e65c5-5aed-4fe9-8414-2092bcb31cc7","shared_citers":4},{"title":"$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization","work_id":"d1ad7304-d09a-49bc-809e-846439f6aff9","shared_citers":3},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":3},{"title":"arXiv preprint arXiv:2111.08897 (2021) 4, 8","work_id":"0ce910be-ca1c-44c7-b7b1-c5353759d85e","shared_citers":3}],"time_series":[{"n":36,"year":2026}],"dependency_candidates":[]},"authors":[]}}