{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CJMQTDXHG2COLKMZUKWBJ6L3BV","short_pith_number":"pith:CJMQTDXH","schema_version":"1.0","canonical_sha256":"1259098ee73684e5a999a2ac14f97b0d73984a2a753131b8d23eb3a65b60cdf1","source":{"kind":"arxiv","id":"2404.19221","version":1},"attestation_state":"computed","paper":{"title":"Transcrib3D: 3D Referring Expression Resolution through Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Gregory Shakhnarovich, Hongyuan Mei, Igor Vasiljevic, Jiading Fang, Matthew R Walter, Rares Ambrus, Shengjie Lin, Vitor Guizilini, Xiangshan Tan","submitted_at":"2024-04-30T02:48:20Z","abstract_excerpt":"If robots are to work effectively alongside people, they must be able to interpret natural language references to objects in their 3D environment. Understanding 3D referring expressions is challenging -- it requires the ability to both parse the 3D structure of the scene and correctly ground free-form language in the presence of distraction and clutter. We introduce Transcrib3D, an approach that brings together 3D detection methods and the emergent reasoning capabilities of large language models (LLMs). Transcrib3D uses text as the unifying medium, which allows us to sidestep the need to learn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.19221","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-30T02:48:20Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e20c9bddcf5d493bce6cc6c812451eba75712b6595f9ca427558bb16e1dc1312","abstract_canon_sha256":"29d1494d070477cf1aede83de0112748792250ec8afee393b831f6e4a7817ade"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:13:40.162945Z","signature_b64":"NvqoYkwd0XR2X3eeUTrh3gYJyy7S9wfT5TzkyB/dhsm5aExQoQGRiHHVjlpHVRQCMWqhk+cWJYtq8A6h6T/3BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1259098ee73684e5a999a2ac14f97b0d73984a2a753131b8d23eb3a65b60cdf1","last_reissued_at":"2026-07-05T08:13:40.162455Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:13:40.162455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transcrib3D: 3D Referring Expression Resolution through Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Gregory Shakhnarovich, Hongyuan Mei, Igor Vasiljevic, Jiading Fang, Matthew R Walter, Rares Ambrus, Shengjie Lin, Vitor Guizilini, Xiangshan Tan","submitted_at":"2024-04-30T02:48:20Z","abstract_excerpt":"If robots are to work effectively alongside people, they must be able to interpret natural language references to objects in their 3D environment. Understanding 3D referring expressions is challenging -- it requires the ability to both parse the 3D structure of the scene and correctly ground free-form language in the presence of distraction and clutter. We introduce Transcrib3D, an approach that brings together 3D detection methods and the emergent reasoning capabilities of large language models (LLMs). Transcrib3D uses text as the unifying medium, which allows us to sidestep the need to learn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.19221","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.19221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.19221","created_at":"2026-07-05T08:13:40.162519+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.19221v1","created_at":"2026-07-05T08:13:40.162519+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.19221","created_at":"2026-07-05T08:13:40.162519+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJMQTDXHG2CO","created_at":"2026-07-05T08:13:40.162519+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJMQTDXHG2COLKMZ","created_at":"2026-07-05T08:13:40.162519+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJMQTDXH","created_at":"2026-07-05T08:13:40.162519+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.19406","citing_title":"MLLM-SUL: Multimodal Large Language Model for Semantic Scene Understanding and Localization in Traffic Scenarios","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV","json":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV.json","graph_json":"https://pith.science/api/pith-number/CJMQTDXHG2COLKMZUKWBJ6L3BV/graph.json","events_json":"https://pith.science/api/pith-number/CJMQTDXHG2COLKMZUKWBJ6L3BV/events.json","paper":"https://pith.science/paper/CJMQTDXH"},"agent_actions":{"view_html":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV","download_json":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV.json","view_paper":"https://pith.science/paper/CJMQTDXH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.19221&json=true","fetch_graph":"https://pith.science/api/pith-number/CJMQTDXHG2COLKMZUKWBJ6L3BV/graph.json","fetch_events":"https://pith.science/api/pith-number/CJMQTDXHG2COLKMZUKWBJ6L3BV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV/action/storage_attestation","attest_author":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV/action/author_attestation","sign_citation":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV/action/citation_signature","submit_replication":"https://pith.science/pith/CJMQTDXHG2COLKMZUKWBJ6L3BV/action/replication_record"}},"created_at":"2026-07-05T08:13:40.162519+00:00","updated_at":"2026-07-05T08:13:40.162519+00:00"}