{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:S6GSHSNWG4P2XIVKMWCUNRA6QE","short_pith_number":"pith:S6GSHSNW","schema_version":"1.0","canonical_sha256":"978d23c9b6371faba2aa658546c41e8128cee9e5b47b64d06506892992c286ac","source":{"kind":"arxiv","id":"2504.18684","version":2},"attestation_state":"computed","paper":{"title":"SORT3D: Spatial Object-centric Reasoning Toolbox for Zero-Shot 3D Grounding Using Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Guofei Chen, Haochen Zhang, Jinkai Qiu, Ji Zhang, Nader Zantout, Pujith Kachana, Wenshan Wang","submitted_at":"2025-04-25T20:24:11Z","abstract_excerpt":"Interpreting object-referential language and grounding objects in 3D with spatial relations and attributes is essential for robots operating alongside humans. However, this task is often challenging due to the diversity of scenes, large number of fine-grained objects, and complex free-form nature of language references. Furthermore, in the 3D domain, obtaining large amounts of natural language training data is difficult. Thus, it is important for methods to learn from little data and zero-shot generalize to new environments. To address these challenges, we propose SORT3D, an approach that util"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.18684","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-25T20:24:11Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"c71454a48bf07f0ec674e4f62fcf9704975caaad55453cd8d1b5e9b81c631efd","abstract_canon_sha256":"8e1a3096e6042779b996e8666c5348ca79f87e855952b04d5520d4b66e03ccc5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:11.740990Z","signature_b64":"mhd8QvCOE8hzIYDfoskylsGIx3JPMhot+sUWV78v+xlHKcOsKDwe+J3oHoYfqbnoVtdbpRvW+GwYxhE6Oe+wDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"978d23c9b6371faba2aa658546c41e8128cee9e5b47b64d06506892992c286ac","last_reissued_at":"2026-07-05T11:54:11.740525Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:11.740525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SORT3D: Spatial Object-centric Reasoning Toolbox for Zero-Shot 3D Grounding Using Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Guofei Chen, Haochen Zhang, Jinkai Qiu, Ji Zhang, Nader Zantout, Pujith Kachana, Wenshan Wang","submitted_at":"2025-04-25T20:24:11Z","abstract_excerpt":"Interpreting object-referential language and grounding objects in 3D with spatial relations and attributes is essential for robots operating alongside humans. However, this task is often challenging due to the diversity of scenes, large number of fine-grained objects, and complex free-form nature of language references. Furthermore, in the 3D domain, obtaining large amounts of natural language training data is difficult. Thus, it is important for methods to learn from little data and zero-shot generalize to new environments. To address these challenges, we propose SORT3D, an approach that util"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.18684","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.18684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.18684","created_at":"2026-07-05T11:54:11.740583+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.18684v2","created_at":"2026-07-05T11:54:11.740583+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.18684","created_at":"2026-07-05T11:54:11.740583+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6GSHSNWG4P2","created_at":"2026-07-05T11:54:11.740583+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6GSHSNWG4P2XIVK","created_at":"2026-07-05T11:54:11.740583+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6GSHSNW","created_at":"2026-07-05T11:54:11.740583+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31144","citing_title":"A Modular Vision-Language-Action Robotics Framework for Indoor Environments","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28490","citing_title":"SSR3D-LLM: Structured Spatial Reasoning via Latent Steps for Fine-Grained Grounding in Unified 3D-LLMs","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09218","citing_title":"Flame3D: Zero-shot Compositional Reasoning of 3D Scenes with Agentic Language Models","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE","json":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE.json","graph_json":"https://pith.science/api/pith-number/S6GSHSNWG4P2XIVKMWCUNRA6QE/graph.json","events_json":"https://pith.science/api/pith-number/S6GSHSNWG4P2XIVKMWCUNRA6QE/events.json","paper":"https://pith.science/paper/S6GSHSNW"},"agent_actions":{"view_html":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE","download_json":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE.json","view_paper":"https://pith.science/paper/S6GSHSNW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.18684&json=true","fetch_graph":"https://pith.science/api/pith-number/S6GSHSNWG4P2XIVKMWCUNRA6QE/graph.json","fetch_events":"https://pith.science/api/pith-number/S6GSHSNWG4P2XIVKMWCUNRA6QE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE/action/storage_attestation","attest_author":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE/action/author_attestation","sign_citation":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE/action/citation_signature","submit_replication":"https://pith.science/pith/S6GSHSNWG4P2XIVKMWCUNRA6QE/action/replication_record"}},"created_at":"2026-07-05T11:54:11.740583+00:00","updated_at":"2026-07-05T11:54:11.740583+00:00"}