{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FUUMMBBHCNJUM5FYT2HAFN6G7J","short_pith_number":"pith:FUUMMBBH","schema_version":"1.0","canonical_sha256":"2d28c6042713534674b89e8e02b7c6fa7ee9835b9adc208e6a930577ecae560a","source":{"kind":"arxiv","id":"2309.16650","version":1},"attestation_state":"computed","paper":{"title":"ConceptGraphs: Open-Vocabulary 3D Scene Graphs for Perception and Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Aditya Agarwal, Alihusein Kuwajerwala, Antonio Torralba, Bipasha Sen, Celso Miguel de Melo, Chuang Gan, Corban Rivera, Florian Shkurti, Joshua B. Tenenbaum, Kirsty Ellis, Krishna Murthy Jatavallabhula, Liam Paull, Qiao Gu, Rama Chellappa, Sacha Morin, William Paul","submitted_at":"2023-09-28T17:53:38Z","abstract_excerpt":"For robots to perform a wide variety of tasks, they require a 3D representation of the world that is semantically rich, yet compact and efficient for task-driven perception and planning. Recent approaches have attempted to leverage features from large vision-language models to encode semantics in 3D representations. However, these approaches tend to produce maps with per-point feature vectors, which do not scale well in larger environments, nor do they contain semantic spatial relationships between entities in the environment, which are useful for downstream planning. In this work, we propose "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16650","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-09-28T17:53:38Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"c0b69b8fbfda8ba71aca9cd114cfddb6f96140ededb47d07b0b712b2e0030d26","abstract_canon_sha256":"f0758a3fe6b8c2094de2085ca908fd99d97da41b5773121ee57d536867b1212b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:19.528638Z","signature_b64":"InOUAxS4RtVRzDwK++ZNnlr5RPu/KzIla4PvLHFTEYU6wGLOrtFvPfUOS61DNHLoAGsQ4I7FRpFUfMXxfT3xAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d28c6042713534674b89e8e02b7c6fa7ee9835b9adc208e6a930577ecae560a","last_reissued_at":"2026-07-05T06:55:19.528163Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:19.528163Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ConceptGraphs: Open-Vocabulary 3D Scene Graphs for Perception and Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Aditya Agarwal, Alihusein Kuwajerwala, Antonio Torralba, Bipasha Sen, Celso Miguel de Melo, Chuang Gan, Corban Rivera, Florian Shkurti, Joshua B. Tenenbaum, Kirsty Ellis, Krishna Murthy Jatavallabhula, Liam Paull, Qiao Gu, Rama Chellappa, Sacha Morin, William Paul","submitted_at":"2023-09-28T17:53:38Z","abstract_excerpt":"For robots to perform a wide variety of tasks, they require a 3D representation of the world that is semantically rich, yet compact and efficient for task-driven perception and planning. Recent approaches have attempted to leverage features from large vision-language models to encode semantics in 3D representations. However, these approaches tend to produce maps with per-point feature vectors, which do not scale well in larger environments, nor do they contain semantic spatial relationships between entities in the environment, which are useful for downstream planning. In this work, we propose "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16650","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16650","created_at":"2026-07-05T06:55:19.528221+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16650v1","created_at":"2026-07-05T06:55:19.528221+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16650","created_at":"2026-07-05T06:55:19.528221+00:00"},{"alias_kind":"pith_short_12","alias_value":"FUUMMBBHCNJU","created_at":"2026-07-05T06:55:19.528221+00:00"},{"alias_kind":"pith_short_16","alias_value":"FUUMMBBHCNJUM5FY","created_at":"2026-07-05T06:55:19.528221+00:00"},{"alias_kind":"pith_short_8","alias_value":"FUUMMBBH","created_at":"2026-07-05T06:55:19.528221+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20946","citing_title":"Scaling Diverse Language Generation for 3D Visual Grounding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01072","citing_title":"Expanding Spatial and Temporal Context for Robotic Imitation Learning With Scene Graphs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31144","citing_title":"A Modular Vision-Language-Action Robotics Framework for Indoor Environments","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14093","citing_title":"A Survey on Vision-Language-Action Models for Embodied AI","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21788","citing_title":"SceneGraphGrounder: Zero-Shot 3D Visual Grounding via Structured Scene Graph Matching","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03669","citing_title":"FUS3DMaps: Scalable and Accurate Open-Vocabulary Semantic Mapping by 3D Fusion of Voxel- and Instance-Level Layers","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J","json":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J.json","graph_json":"https://pith.science/api/pith-number/FUUMMBBHCNJUM5FYT2HAFN6G7J/graph.json","events_json":"https://pith.science/api/pith-number/FUUMMBBHCNJUM5FYT2HAFN6G7J/events.json","paper":"https://pith.science/paper/FUUMMBBH"},"agent_actions":{"view_html":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J","download_json":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J.json","view_paper":"https://pith.science/paper/FUUMMBBH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16650&json=true","fetch_graph":"https://pith.science/api/pith-number/FUUMMBBHCNJUM5FYT2HAFN6G7J/graph.json","fetch_events":"https://pith.science/api/pith-number/FUUMMBBHCNJUM5FYT2HAFN6G7J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J/action/storage_attestation","attest_author":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J/action/author_attestation","sign_citation":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J/action/citation_signature","submit_replication":"https://pith.science/pith/FUUMMBBHCNJUM5FYT2HAFN6G7J/action/replication_record"}},"created_at":"2026-07-05T06:55:19.528221+00:00","updated_at":"2026-07-05T06:55:19.528221+00:00"}