{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WPPWN65AC7227NKMYXP4NFXHJS","short_pith_number":"pith:WPPWN65A","schema_version":"1.0","canonical_sha256":"b3df66fba017f5afb54cc5dfc696e74c8df9d0a912312d15cead306d9bc9d811","source":{"kind":"arxiv","id":"2406.02548","version":3},"attestation_state":"computed","paper":{"title":"Open-YOLO 3D: Towards Fast and Accurate Open-Vocabulary 3D Instance Segmentation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angela Dai, Fahad Shahbaz Khan, Hisham Cholakkal, Jean Lahoud, Mohamed el amine boudjoghra, Rao Muhammad Anwer, Salman Khan","submitted_at":"2024-06-04T17:59:31Z","abstract_excerpt":"Recent works on open-vocabulary 3D instance segmentation show strong promise, but at the cost of slow inference speed and high computation requirements. This high computation cost is typically due to their heavy reliance on 3D clip features, which require computationally expensive 2D foundation models like Segment Anything (SAM) and CLIP for multi-view aggregation into 3D. As a consequence, this hampers their applicability in many real-world applications that require both fast and accurate predictions. To this end, we propose a fast yet accurate open-vocabulary 3D instance segmentation approac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02548","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-04T17:59:31Z","cross_cats_sorted":[],"title_canon_sha256":"225b159b044572412bc2951928a3f9ff6b736472a2a61fd4579f3b3b63c1214f","abstract_canon_sha256":"1e7c88ff5ef99b57160cf2b70fc9e76485427da241853be05844e8f9b48665c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:40.468800Z","signature_b64":"2zW1P1wASfnPZ2aCWjJ6RcygyUqs4QSXHDJPpIoig+YMCGveezSJcCwiqCzhfq/sTlJHaxP5Db5wp/xCcuf5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b3df66fba017f5afb54cc5dfc696e74c8df9d0a912312d15cead306d9bc9d811","last_reissued_at":"2026-07-05T10:13:40.468356Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:40.468356Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Open-YOLO 3D: Towards Fast and Accurate Open-Vocabulary 3D Instance Segmentation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Angela Dai, Fahad Shahbaz Khan, Hisham Cholakkal, Jean Lahoud, Mohamed el amine boudjoghra, Rao Muhammad Anwer, Salman Khan","submitted_at":"2024-06-04T17:59:31Z","abstract_excerpt":"Recent works on open-vocabulary 3D instance segmentation show strong promise, but at the cost of slow inference speed and high computation requirements. This high computation cost is typically due to their heavy reliance on 3D clip features, which require computationally expensive 2D foundation models like Segment Anything (SAM) and CLIP for multi-view aggregation into 3D. As a consequence, this hampers their applicability in many real-world applications that require both fast and accurate predictions. To this end, we propose a fast yet accurate open-vocabulary 3D instance segmentation approac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02548","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02548/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02548","created_at":"2026-07-05T10:13:40.468444+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02548v3","created_at":"2026-07-05T10:13:40.468444+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02548","created_at":"2026-07-05T10:13:40.468444+00:00"},{"alias_kind":"pith_short_12","alias_value":"WPPWN65AC722","created_at":"2026-07-05T10:13:40.468444+00:00"},{"alias_kind":"pith_short_16","alias_value":"WPPWN65AC7227NKM","created_at":"2026-07-05T10:13:40.468444+00:00"},{"alias_kind":"pith_short_8","alias_value":"WPPWN65A","created_at":"2026-07-05T10:13:40.468444+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30638","citing_title":"Open-Vocabulary and Referring Segmentation for 3D Gaussians Using 2D Detectors","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03532","citing_title":"OpenTrack3D: Towards Accurate and Generalizable Open-Vocabulary 3D Instance Segmentation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2601.08831","citing_title":"3AM: 3egment Anything with Geometric Consistency in Videos","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS","json":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS.json","graph_json":"https://pith.science/api/pith-number/WPPWN65AC7227NKMYXP4NFXHJS/graph.json","events_json":"https://pith.science/api/pith-number/WPPWN65AC7227NKMYXP4NFXHJS/events.json","paper":"https://pith.science/paper/WPPWN65A"},"agent_actions":{"view_html":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS","download_json":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS.json","view_paper":"https://pith.science/paper/WPPWN65A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02548&json=true","fetch_graph":"https://pith.science/api/pith-number/WPPWN65AC7227NKMYXP4NFXHJS/graph.json","fetch_events":"https://pith.science/api/pith-number/WPPWN65AC7227NKMYXP4NFXHJS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS/action/storage_attestation","attest_author":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS/action/author_attestation","sign_citation":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS/action/citation_signature","submit_replication":"https://pith.science/pith/WPPWN65AC7227NKMYXP4NFXHJS/action/replication_record"}},"created_at":"2026-07-05T10:13:40.468444+00:00","updated_at":"2026-07-05T10:13:40.468444+00:00"}