{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z7JP7OXDJSB6DTTMRFRDKYKC3L","short_pith_number":"pith:Z7JP7OXD","schema_version":"1.0","canonical_sha256":"cfd2ffbae34c83e1ce6c8962356142daf95e66f9e3b172c973337405f096c845","source":{"kind":"arxiv","id":"2506.05318","version":2},"attestation_state":"computed","paper":{"title":"Does Your 3D Encoder Really Work? When Pretrain-SFT from 2D VLMs Meets 3D VLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dave Zhenyu Chen, Hang Xu, Haoyuan Li, Jianhua Han, JiaWang Bian, Tao Tang, Xiaodan Liang, Yanpeng Zhou, Yufei Gao, Yujie Yuan","submitted_at":"2025-06-05T17:56:12Z","abstract_excerpt":"Remarkable progress in 2D Vision-Language Models (VLMs) has spurred interest in extending them to 3D settings for tasks like 3D Question Answering, Dense Captioning, and Visual Grounding. Unlike 2D VLMs that typically process images through an image encoder, 3D scenes, with their intricate spatial structures, allow for diverse model architectures. Based on their encoder design, this paper categorizes recent 3D VLMs into 3D object-centric, 2D image-based, and 3D scene-centric approaches. Despite the architectural similarity of 3D scene-centric VLMs to their 2D counterparts, they have exhibited "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.05318","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-05T17:56:12Z","cross_cats_sorted":[],"title_canon_sha256":"cd8cd385de1fdc9dc86421d34e06d72d05f715c9d1bd3c01ab315b1ddaf7f1cf","abstract_canon_sha256":"a03e98305cf88342173f2453c959d52548afe69d5552c8639cf45bb624ea3925"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:01.005325Z","signature_b64":"Q4IpcJvlwLMJTo5ERM5lpBM1GMq4fQAhb876MqT+u+k7xDHLWvZslwDJYMPllD6l8UvEmiJrTgqbHSp+mtnkBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cfd2ffbae34c83e1ce6c8962356142daf95e66f9e3b172c973337405f096c845","last_reissued_at":"2026-07-05T11:17:01.004752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:01.004752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Your 3D Encoder Really Work? When Pretrain-SFT from 2D VLMs Meets 3D VLMs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dave Zhenyu Chen, Hang Xu, Haoyuan Li, Jianhua Han, JiaWang Bian, Tao Tang, Xiaodan Liang, Yanpeng Zhou, Yufei Gao, Yujie Yuan","submitted_at":"2025-06-05T17:56:12Z","abstract_excerpt":"Remarkable progress in 2D Vision-Language Models (VLMs) has spurred interest in extending them to 3D settings for tasks like 3D Question Answering, Dense Captioning, and Visual Grounding. Unlike 2D VLMs that typically process images through an image encoder, 3D scenes, with their intricate spatial structures, allow for diverse model architectures. Based on their encoder design, this paper categorizes recent 3D VLMs into 3D object-centric, 2D image-based, and 3D scene-centric approaches. Despite the architectural similarity of 3D scene-centric VLMs to their 2D counterparts, they have exhibited "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.05318","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.05318/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.05318","created_at":"2026-07-05T11:17:01.004825+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.05318v2","created_at":"2026-07-05T11:17:01.004825+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.05318","created_at":"2026-07-05T11:17:01.004825+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z7JP7OXDJSB6","created_at":"2026-07-05T11:17:01.004825+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z7JP7OXDJSB6DTTM","created_at":"2026-07-05T11:17:01.004825+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z7JP7OXD","created_at":"2026-07-05T11:17:01.004825+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.12908","citing_title":"Robotic Manipulation is Vision-to-Geometry Mapping ($f(v) \\rightarrow G$): Vision-Geometry Backbones over Language and Video Models","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L","json":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L.json","graph_json":"https://pith.science/api/pith-number/Z7JP7OXDJSB6DTTMRFRDKYKC3L/graph.json","events_json":"https://pith.science/api/pith-number/Z7JP7OXDJSB6DTTMRFRDKYKC3L/events.json","paper":"https://pith.science/paper/Z7JP7OXD"},"agent_actions":{"view_html":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L","download_json":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L.json","view_paper":"https://pith.science/paper/Z7JP7OXD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.05318&json=true","fetch_graph":"https://pith.science/api/pith-number/Z7JP7OXDJSB6DTTMRFRDKYKC3L/graph.json","fetch_events":"https://pith.science/api/pith-number/Z7JP7OXDJSB6DTTMRFRDKYKC3L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L/action/storage_attestation","attest_author":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L/action/author_attestation","sign_citation":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L/action/citation_signature","submit_replication":"https://pith.science/pith/Z7JP7OXDJSB6DTTMRFRDKYKC3L/action/replication_record"}},"created_at":"2026-07-05T11:17:01.004825+00:00","updated_at":"2026-07-05T11:17:01.004825+00:00"}