{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4YOYD5JF6FTZDKFNIWE73CWNP2","short_pith_number":"pith:4YOYD5JF","schema_version":"1.0","canonical_sha256":"e61d81f525f16791a8ad4589fd8acd7eb88d69c10f1aa266497a42446e56b313","source":{"kind":"arxiv","id":"2406.05785","version":2},"attestation_state":"computed","paper":{"title":"A Survey on Text-guided 3D Visual Grounding: Elements, Recent Advances, and Future Directions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daizong Liu, Wei Hu, Wencan Huang, Yang Liu","submitted_at":"2024-06-09T13:52:12Z","abstract_excerpt":"Text-guided 3D visual grounding (T-3DVG), which aims to locate a specific object that semantically corresponds to a language query from a complicated 3D scene, has drawn increasing attention in the 3D research community over the past few years. Compared to 2D visual grounding, this task presents great potential and challenges due to its closer proximity to the real world and the complexity of data collection and 3D point cloud source processing. In this survey, we attempt to provide a comprehensive overview of the T-3DVG progress, including its fundamental elements, recent research advances, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05785","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-09T13:52:12Z","cross_cats_sorted":[],"title_canon_sha256":"ca9fd12280d1413022c68c93a304c279352113d0170b26cc0c13d4d13412e7b7","abstract_canon_sha256":"91e32070a02c65c7db959d2c35701cd6123d608c9deea66cce77daf1e627a903"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:25.700538Z","signature_b64":"aaZHb4FM3mKvOXDQekrkYIqGFaIH4lkpu0oluTMd6AgvLCM9ENA1/PtNQz9M2QnbSH7UBoXhTgUKkGJ7ykcXDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e61d81f525f16791a8ad4589fd8acd7eb88d69c10f1aa266497a42446e56b313","last_reissued_at":"2026-07-05T08:46:25.700056Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:25.700056Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Text-guided 3D Visual Grounding: Elements, Recent Advances, and Future Directions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daizong Liu, Wei Hu, Wencan Huang, Yang Liu","submitted_at":"2024-06-09T13:52:12Z","abstract_excerpt":"Text-guided 3D visual grounding (T-3DVG), which aims to locate a specific object that semantically corresponds to a language query from a complicated 3D scene, has drawn increasing attention in the 3D research community over the past few years. Compared to 2D visual grounding, this task presents great potential and challenges due to its closer proximity to the real world and the complexity of data collection and 3D point cloud source processing. In this survey, we attempt to provide a comprehensive overview of the T-3DVG progress, including its fundamental elements, recent research advances, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05785","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05785/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05785","created_at":"2026-07-05T08:46:25.700113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05785v2","created_at":"2026-07-05T08:46:25.700113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05785","created_at":"2026-07-05T08:46:25.700113+00:00"},{"alias_kind":"pith_short_12","alias_value":"4YOYD5JF6FTZ","created_at":"2026-07-05T08:46:25.700113+00:00"},{"alias_kind":"pith_short_16","alias_value":"4YOYD5JF6FTZDKFN","created_at":"2026-07-05T08:46:25.700113+00:00"},{"alias_kind":"pith_short_8","alias_value":"4YOYD5JF","created_at":"2026-07-05T08:46:25.700113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28490","citing_title":"SSR3D-LLM: Structured Spatial Reasoning via Latent Steps for Fine-Grained Grounding in Unified 3D-LLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2412.00404","citing_title":"Hard-Label Black-Box Attacks on 3D Point Clouds","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02689","citing_title":"Efficient3D: A Unified Framework for Adaptive and Debiased Token Reduction in 3D MLLMs","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2","json":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2.json","graph_json":"https://pith.science/api/pith-number/4YOYD5JF6FTZDKFNIWE73CWNP2/graph.json","events_json":"https://pith.science/api/pith-number/4YOYD5JF6FTZDKFNIWE73CWNP2/events.json","paper":"https://pith.science/paper/4YOYD5JF"},"agent_actions":{"view_html":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2","download_json":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2.json","view_paper":"https://pith.science/paper/4YOYD5JF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05785&json=true","fetch_graph":"https://pith.science/api/pith-number/4YOYD5JF6FTZDKFNIWE73CWNP2/graph.json","fetch_events":"https://pith.science/api/pith-number/4YOYD5JF6FTZDKFNIWE73CWNP2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2/action/storage_attestation","attest_author":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2/action/author_attestation","sign_citation":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2/action/citation_signature","submit_replication":"https://pith.science/pith/4YOYD5JF6FTZDKFNIWE73CWNP2/action/replication_record"}},"created_at":"2026-07-05T08:46:25.700113+00:00","updated_at":"2026-07-05T08:46:25.700113+00:00"}