{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N4TVG4RGV7SN6F2XKMFGIUUIZY","short_pith_number":"pith:N4TVG4RG","schema_version":"1.0","canonical_sha256":"6f27537226afe4df1757530a645288ce0a7bae5633fb760101151d2e6e973759","source":{"kind":"arxiv","id":"2507.04047","version":2},"attestation_state":"computed","paper":{"title":"Move to Understand a 3D Scene: Bridging Visual Grounding and Exploration for Efficient and Versatile Embodied Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoxiong Jia, Qian Yu, Qing Li, Siyuan Huang, Wei Liang, Xiaojian Ma, Xilin Wang, Yixin Chen, Yixuan Li, Zhidong Deng, Zhuofan Zhang, Ziyu Zhu","submitted_at":"2025-07-05T14:15:52Z","abstract_excerpt":"Embodied scene understanding requires not only comprehending visual-spatial information that has been observed but also determining where to explore next in the 3D physical world. Existing 3D Vision-Language (3D-VL) models primarily focus on grounding objects in static observations from 3D reconstruction, such as meshes and point clouds, but lack the ability to actively perceive and explore their environment. To address this limitation, we introduce \\underline{\\textbf{M}}ove \\underline{\\textbf{t}}o \\underline{\\textbf{U}}nderstand (\\textbf{\\model}), a unified framework that integrates active pe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.04047","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-05T14:15:52Z","cross_cats_sorted":[],"title_canon_sha256":"9c77a29a71238f42eeb9a66209b747c79b7d14dcb94e50f4f10fc6732635b15d","abstract_canon_sha256":"53eb9eb1f7940341eec01a06b093a1552a2ce833c30d4ba0159130a2efed5d7e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:37.295071Z","signature_b64":"hAG2kGVRe9Wvh+7fFlZFYHuY/F1+hXE2KfUuIE/pmjDqqr/APa1IR6GNH05xzVjLigiPnRprJNSMM86oE4OwDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f27537226afe4df1757530a645288ce0a7bae5633fb760101151d2e6e973759","last_reissued_at":"2026-07-05T11:45:37.294572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:37.294572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Move to Understand a 3D Scene: Bridging Visual Grounding and Exploration for Efficient and Versatile Embodied Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baoxiong Jia, Qian Yu, Qing Li, Siyuan Huang, Wei Liang, Xiaojian Ma, Xilin Wang, Yixin Chen, Yixuan Li, Zhidong Deng, Zhuofan Zhang, Ziyu Zhu","submitted_at":"2025-07-05T14:15:52Z","abstract_excerpt":"Embodied scene understanding requires not only comprehending visual-spatial information that has been observed but also determining where to explore next in the 3D physical world. Existing 3D Vision-Language (3D-VL) models primarily focus on grounding objects in static observations from 3D reconstruction, such as meshes and point clouds, but lack the ability to actively perceive and explore their environment. To address this limitation, we introduce \\underline{\\textbf{M}}ove \\underline{\\textbf{t}}o \\underline{\\textbf{U}}nderstand (\\textbf{\\model}), a unified framework that integrates active pe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.04047","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.04047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.04047","created_at":"2026-07-05T11:45:37.294632+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.04047v2","created_at":"2026-07-05T11:45:37.294632+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.04047","created_at":"2026-07-05T11:45:37.294632+00:00"},{"alias_kind":"pith_short_12","alias_value":"N4TVG4RGV7SN","created_at":"2026-07-05T11:45:37.294632+00:00"},{"alias_kind":"pith_short_16","alias_value":"N4TVG4RGV7SN6F2X","created_at":"2026-07-05T11:45:37.294632+00:00"},{"alias_kind":"pith_short_8","alias_value":"N4TVG4RG","created_at":"2026-07-05T11:45:37.294632+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.03139","citing_title":"FSUNav: A Cerebrum-Cerebellum Architecture for Fast, Safe, and Universal Zero-Shot Goal-Oriented Navigation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY","json":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY.json","graph_json":"https://pith.science/api/pith-number/N4TVG4RGV7SN6F2XKMFGIUUIZY/graph.json","events_json":"https://pith.science/api/pith-number/N4TVG4RGV7SN6F2XKMFGIUUIZY/events.json","paper":"https://pith.science/paper/N4TVG4RG"},"agent_actions":{"view_html":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY","download_json":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY.json","view_paper":"https://pith.science/paper/N4TVG4RG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.04047&json=true","fetch_graph":"https://pith.science/api/pith-number/N4TVG4RGV7SN6F2XKMFGIUUIZY/graph.json","fetch_events":"https://pith.science/api/pith-number/N4TVG4RGV7SN6F2XKMFGIUUIZY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY/action/storage_attestation","attest_author":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY/action/author_attestation","sign_citation":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY/action/citation_signature","submit_replication":"https://pith.science/pith/N4TVG4RGV7SN6F2XKMFGIUUIZY/action/replication_record"}},"created_at":"2026-07-05T11:45:37.294632+00:00","updated_at":"2026-07-05T11:45:37.294632+00:00"}