{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VCFY375RF342UY27XFXWZMWW5K","short_pith_number":"pith:VCFY375R","schema_version":"1.0","canonical_sha256":"a88b8dffb12ef9aa635fb96f6cb2d6eab3bdd6b4159d21041958205cc9970903","source":{"kind":"arxiv","id":"2309.09456","version":1},"attestation_state":"computed","paper":{"title":"Object2Scene: Putting Objects in Context for Open-Vocabulary 3D Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenming Zhu, Kai Chen, Tai Wang, Wenwei Zhang, Xihui Liu","submitted_at":"2023-09-18T03:31:53Z","abstract_excerpt":"Point cloud-based open-vocabulary 3D object detection aims to detect 3D categories that do not have ground-truth annotations in the training set. It is extremely challenging because of the limited data and annotations (bounding boxes with class labels or text descriptions) of 3D scenes. Previous approaches leverage large-scale richly-annotated image datasets as a bridge between 3D and category semantics but require an extra alignment process between 2D images and 3D points, limiting the open-vocabulary ability of 3D detectors. Instead of leveraging 2D images, we propose Object2Scene, the first"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.09456","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-18T03:31:53Z","cross_cats_sorted":[],"title_canon_sha256":"2c79d02b9ca4a5856b196ac4dbccbcf3cd5b7f972a4a2a4efed0c42a9321bb7d","abstract_canon_sha256":"743ae82b1f1cb4cbb9eadf35e7fde45b07ef17d987d24000d49e962ab4498356"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:51:39.291367Z","signature_b64":"C9YfgIUGcadIUhD8uEsIiF4QpX7p/qFicSbBMUH1B2PN5DE/Yow2lZU1sVq5BFAYBZThgVVqCD43R+cfrpA7Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a88b8dffb12ef9aa635fb96f6cb2d6eab3bdd6b4159d21041958205cc9970903","last_reissued_at":"2026-07-05T06:51:39.290824Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:51:39.290824Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Object2Scene: Putting Objects in Context for Open-Vocabulary 3D Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenming Zhu, Kai Chen, Tai Wang, Wenwei Zhang, Xihui Liu","submitted_at":"2023-09-18T03:31:53Z","abstract_excerpt":"Point cloud-based open-vocabulary 3D object detection aims to detect 3D categories that do not have ground-truth annotations in the training set. It is extremely challenging because of the limited data and annotations (bounding boxes with class labels or text descriptions) of 3D scenes. Previous approaches leverage large-scale richly-annotated image datasets as a bridge between 3D and category semantics but require an extra alignment process between 2D images and 3D points, limiting the open-vocabulary ability of 3D detectors. Instead of leveraging 2D images, we propose Object2Scene, the first"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.09456","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.09456/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.09456","created_at":"2026-07-05T06:51:39.290895+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.09456v1","created_at":"2026-07-05T06:51:39.290895+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.09456","created_at":"2026-07-05T06:51:39.290895+00:00"},{"alias_kind":"pith_short_12","alias_value":"VCFY375RF342","created_at":"2026-07-05T06:51:39.290895+00:00"},{"alias_kind":"pith_short_16","alias_value":"VCFY375RF342UY27","created_at":"2026-07-05T06:51:39.290895+00:00"},{"alias_kind":"pith_short_8","alias_value":"VCFY375R","created_at":"2026-07-05T06:51:39.290895+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24805","citing_title":"DDStereo: Efficient Dual Decoder Transformers for Stereo 3D Road Anomaly Detection","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10573","citing_title":"Learning 3D Representations for Spatial Intelligence from Unposed Multi-View Images","ref_index":89,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K","json":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K.json","graph_json":"https://pith.science/api/pith-number/VCFY375RF342UY27XFXWZMWW5K/graph.json","events_json":"https://pith.science/api/pith-number/VCFY375RF342UY27XFXWZMWW5K/events.json","paper":"https://pith.science/paper/VCFY375R"},"agent_actions":{"view_html":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K","download_json":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K.json","view_paper":"https://pith.science/paper/VCFY375R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.09456&json=true","fetch_graph":"https://pith.science/api/pith-number/VCFY375RF342UY27XFXWZMWW5K/graph.json","fetch_events":"https://pith.science/api/pith-number/VCFY375RF342UY27XFXWZMWW5K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K/action/storage_attestation","attest_author":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K/action/author_attestation","sign_citation":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K/action/citation_signature","submit_replication":"https://pith.science/pith/VCFY375RF342UY27XFXWZMWW5K/action/replication_record"}},"created_at":"2026-07-05T06:51:39.290895+00:00","updated_at":"2026-07-05T06:51:39.290895+00:00"}