{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SJBMYILQAS26REMJUZH2NBWI2R","short_pith_number":"pith:SJBMYILQ","schema_version":"1.0","canonical_sha256":"9242cc217004b5e89189a64fa686c8d4722e1aeb0fc8bc917635f1135dea92ea","source":{"kind":"arxiv","id":"2312.16170","version":1},"attestation_state":"computed","paper":{"title":"EmbodiedScan: A Holistic Multi-Modal 3D Perception Suite Towards Embodied AI","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Cewu Lu, Chenming Zhu, Dahua Lin, Jiangmiao Pang, Kai Chen, Peisen Li, Ruiyuan Lyu, Runsen Xu, Tai Wang, Tianfan Xue, Wenwei Zhang, Xiao Chen, Xiaohan Mao, Xihui Liu","submitted_at":"2023-12-26T18:59:11Z","abstract_excerpt":"In the realm of computer vision and robotics, embodied agents are expected to explore their environment and carry out human instructions. This necessitates the ability to fully understand 3D scenes given their first-person observations and contextualize them into language for interaction. However, traditional research focuses more on scene-level input and output setups from a global view. To address the gap, we introduce EmbodiedScan, a multi-modal, ego-centric 3D perception dataset and benchmark for holistic 3D scene understanding. It encompasses over 5k scans encapsulating 1M ego-centric RGB"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.16170","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-12-26T18:59:11Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"6661b89a7f33876e5c1971507d5ea84a7fb4a2491264ea64e81ba1d86a1cc90c","abstract_canon_sha256":"db25e90d0c4491a96be67085239388cd7de190d29ee28fb1afb522e1578e823d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:06.915940Z","signature_b64":"QkV5KK1hXKTfYsXjGNyDWrLuusvT+9vlM0pCC1IR4TctbAvkv+jQe6evyvyltVSLbKrxZurIeOWOG4dOFPt/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9242cc217004b5e89189a64fa686c8d4722e1aeb0fc8bc917635f1135dea92ea","last_reissued_at":"2026-07-05T07:28:06.915386Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:06.915386Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EmbodiedScan: A Holistic Multi-Modal 3D Perception Suite Towards Embodied AI","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Cewu Lu, Chenming Zhu, Dahua Lin, Jiangmiao Pang, Kai Chen, Peisen Li, Ruiyuan Lyu, Runsen Xu, Tai Wang, Tianfan Xue, Wenwei Zhang, Xiao Chen, Xiaohan Mao, Xihui Liu","submitted_at":"2023-12-26T18:59:11Z","abstract_excerpt":"In the realm of computer vision and robotics, embodied agents are expected to explore their environment and carry out human instructions. This necessitates the ability to fully understand 3D scenes given their first-person observations and contextualize them into language for interaction. However, traditional research focuses more on scene-level input and output setups from a global view. To address the gap, we introduce EmbodiedScan, a multi-modal, ego-centric 3D perception dataset and benchmark for holistic 3D scene understanding. It encompasses over 5k scans encapsulating 1M ego-centric RGB"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.16170","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.16170/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.16170","created_at":"2026-07-05T07:28:06.915448+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.16170v1","created_at":"2026-07-05T07:28:06.915448+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.16170","created_at":"2026-07-05T07:28:06.915448+00:00"},{"alias_kind":"pith_short_12","alias_value":"SJBMYILQAS26","created_at":"2026-07-05T07:28:06.915448+00:00"},{"alias_kind":"pith_short_16","alias_value":"SJBMYILQAS26REMJ","created_at":"2026-07-05T07:28:06.915448+00:00"},{"alias_kind":"pith_short_8","alias_value":"SJBMYILQ","created_at":"2026-07-05T07:28:06.915448+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22971","citing_title":"Humanoid-OmniOcc: Stereo-Based Full-View Occupancy Dataset for Embodied AI","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R","json":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R.json","graph_json":"https://pith.science/api/pith-number/SJBMYILQAS26REMJUZH2NBWI2R/graph.json","events_json":"https://pith.science/api/pith-number/SJBMYILQAS26REMJUZH2NBWI2R/events.json","paper":"https://pith.science/paper/SJBMYILQ"},"agent_actions":{"view_html":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R","download_json":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R.json","view_paper":"https://pith.science/paper/SJBMYILQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.16170&json=true","fetch_graph":"https://pith.science/api/pith-number/SJBMYILQAS26REMJUZH2NBWI2R/graph.json","fetch_events":"https://pith.science/api/pith-number/SJBMYILQAS26REMJUZH2NBWI2R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R/action/storage_attestation","attest_author":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R/action/author_attestation","sign_citation":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R/action/citation_signature","submit_replication":"https://pith.science/pith/SJBMYILQAS26REMJUZH2NBWI2R/action/replication_record"}},"created_at":"2026-07-05T07:28:06.915448+00:00","updated_at":"2026-07-05T07:28:06.915448+00:00"}