{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OAIMX2KWE7VOK3LIXXRTXNLFYD","short_pith_number":"pith:OAIMX2KW","schema_version":"1.0","canonical_sha256":"7010cbe95627eae56d68bde33bb565c0d03d54d3cdc2a2799bbd68cc3d87ba1f","source":{"kind":"arxiv","id":"2408.11811","version":3},"attestation_state":"computed","paper":{"title":"EmbodiedSAM: Online Segment Any 3D Thing in Real Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Huangxing Chen, Jie Zhou, Jiwen Lu, Linqing Zhao, Xiuwei Xu, Ziwei Wang","submitted_at":"2024-08-21T17:57:06Z","abstract_excerpt":"Embodied tasks require the agent to fully understand 3D scenes simultaneously with its exploration, so an online, real-time, fine-grained and highly-generalized 3D perception model is desperately needed. Since high-quality 3D data is limited, directly training such a model in 3D is almost infeasible. Meanwhile, vision foundation models (VFM) has revolutionized the field of 2D computer vision with superior performance, which makes the use of VFM to assist embodied 3D perception a promising direction. However, most existing VFM-assisted 3D perception methods are either offline or too slow that c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11811","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-21T17:57:06Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"32fcbc62df67d643dc14a08940aa560a9569408ce3526e05975a0e5e538add8e","abstract_canon_sha256":"8b075c01f9918feb192833221e9ec21965aeca48aa149de57c481a951bcdb330"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:02.865744Z","signature_b64":"wFrdCrfUl4srYt0pHFPCdL/PpOTaXGMtdM9cFEcdA4LD8hWEkP6X2RH1RCPE6aXvoHzxPgkF1L2GJB7+rfcKCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7010cbe95627eae56d68bde33bb565c0d03d54d3cdc2a2799bbd68cc3d87ba1f","last_reissued_at":"2026-07-05T10:13:02.865150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:02.865150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EmbodiedSAM: Online Segment Any 3D Thing in Real Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Huangxing Chen, Jie Zhou, Jiwen Lu, Linqing Zhao, Xiuwei Xu, Ziwei Wang","submitted_at":"2024-08-21T17:57:06Z","abstract_excerpt":"Embodied tasks require the agent to fully understand 3D scenes simultaneously with its exploration, so an online, real-time, fine-grained and highly-generalized 3D perception model is desperately needed. Since high-quality 3D data is limited, directly training such a model in 3D is almost infeasible. Meanwhile, vision foundation models (VFM) has revolutionized the field of 2D computer vision with superior performance, which makes the use of VFM to assist embodied 3D perception a promising direction. However, most existing VFM-assisted 3D perception methods are either offline or too slow that c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11811","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11811","created_at":"2026-07-05T10:13:02.865222+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11811v3","created_at":"2026-07-05T10:13:02.865222+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11811","created_at":"2026-07-05T10:13:02.865222+00:00"},{"alias_kind":"pith_short_12","alias_value":"OAIMX2KWE7VO","created_at":"2026-07-05T10:13:02.865222+00:00"},{"alias_kind":"pith_short_16","alias_value":"OAIMX2KWE7VOK3LI","created_at":"2026-07-05T10:13:02.865222+00:00"},{"alias_kind":"pith_short_8","alias_value":"OAIMX2KW","created_at":"2026-07-05T10:13:02.865222+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17520","citing_title":"GASE: Gaussian Splatting-Based Automated System for Reconstructing Embodied-Simulation Environments","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12849","citing_title":"SemanticXR: Low Power and Real-time Queryable Semantic Mapping with an Object-Level Device-Cloud Architecture","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14801","citing_title":"Exploring Bottlenecks in VLM-LLM Navigation: How 3D Scene Understanding Capability Impacts Zero-Shot VLN","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29505","citing_title":"ESAM++: Efficient Online 3D Perception on the Edge","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2601.08831","citing_title":"3AM: 3egment Anything with Geometric Consistency in Videos","ref_index":95,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD","json":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD.json","graph_json":"https://pith.science/api/pith-number/OAIMX2KWE7VOK3LIXXRTXNLFYD/graph.json","events_json":"https://pith.science/api/pith-number/OAIMX2KWE7VOK3LIXXRTXNLFYD/events.json","paper":"https://pith.science/paper/OAIMX2KW"},"agent_actions":{"view_html":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD","download_json":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD.json","view_paper":"https://pith.science/paper/OAIMX2KW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11811&json=true","fetch_graph":"https://pith.science/api/pith-number/OAIMX2KWE7VOK3LIXXRTXNLFYD/graph.json","fetch_events":"https://pith.science/api/pith-number/OAIMX2KWE7VOK3LIXXRTXNLFYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD/action/storage_attestation","attest_author":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD/action/author_attestation","sign_citation":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD/action/citation_signature","submit_replication":"https://pith.science/pith/OAIMX2KWE7VOK3LIXXRTXNLFYD/action/replication_record"}},"created_at":"2026-07-05T10:13:02.865222+00:00","updated_at":"2026-07-05T10:13:02.865222+00:00"}