{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QBMRU7DWBGWVPD4LDWASC3N4AY","short_pith_number":"pith:QBMRU7DW","schema_version":"1.0","canonical_sha256":"80591a7c7609ad578f8b1d81216dbc062a09a3c2a4d79cab1ba95ac4f4474cbd","source":{"kind":"arxiv","id":"2403.12396","version":1},"attestation_state":"computed","paper":{"title":"OV9D: Open-Vocabulary Category-Level 9D Object Pose and Size Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Junhao Cai, Liefeng Bo, Qifeng Chen, Siyu Zhu, Weihao Yuan, Yisheng He, Zilong Dong","submitted_at":"2024-03-19T03:09:24Z","abstract_excerpt":"This paper studies a new open-set problem, the open-vocabulary category-level object pose and size estimation. Given human text descriptions of arbitrary novel object categories, the robot agent seeks to predict the position, orientation, and size of the target object in the observed scene image. To enable such generalizability, we first introduce OO3D-9D, a large-scale photorealistic dataset for this task. Derived from OmniObject3D, OO3D-9D is the largest and most diverse dataset in the field of category-level object pose and size estimation. It includes additional annotations for the symmetr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.12396","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-19T03:09:24Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"8361026a3b05b2eb2031ebcdd2ff87e7e4a69e603cdae33f2863ace93af9ff52","abstract_canon_sha256":"96922c362a920f278adebb117a3d39907932b89021972421e652b730369f1b2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:58:00.789702Z","signature_b64":"wNgWBlD/onm6fKmMwyKI5YoGfJGec+Q/dnFM+DipdPNKgcws+WYjNs71WILfzl1QJv+QWz0WHnwNlgNPCuwZAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80591a7c7609ad578f8b1d81216dbc062a09a3c2a4d79cab1ba95ac4f4474cbd","last_reissued_at":"2026-07-05T07:58:00.789275Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:58:00.789275Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OV9D: Open-Vocabulary Category-Level 9D Object Pose and Size Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Junhao Cai, Liefeng Bo, Qifeng Chen, Siyu Zhu, Weihao Yuan, Yisheng He, Zilong Dong","submitted_at":"2024-03-19T03:09:24Z","abstract_excerpt":"This paper studies a new open-set problem, the open-vocabulary category-level object pose and size estimation. Given human text descriptions of arbitrary novel object categories, the robot agent seeks to predict the position, orientation, and size of the target object in the observed scene image. To enable such generalizability, we first introduce OO3D-9D, a large-scale photorealistic dataset for this task. Derived from OmniObject3D, OO3D-9D is the largest and most diverse dataset in the field of category-level object pose and size estimation. It includes additional annotations for the symmetr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.12396","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.12396/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.12396","created_at":"2026-07-05T07:58:00.789345+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.12396v1","created_at":"2026-07-05T07:58:00.789345+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.12396","created_at":"2026-07-05T07:58:00.789345+00:00"},{"alias_kind":"pith_short_12","alias_value":"QBMRU7DWBGWV","created_at":"2026-07-05T07:58:00.789345+00:00"},{"alias_kind":"pith_short_16","alias_value":"QBMRU7DWBGWVPD4L","created_at":"2026-07-05T07:58:00.789345+00:00"},{"alias_kind":"pith_short_8","alias_value":"QBMRU7DW","created_at":"2026-07-05T07:58:00.789345+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30058","citing_title":"Emergence of a Shared Canonical Object Frame from In-the-Wild Videos","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY","json":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY.json","graph_json":"https://pith.science/api/pith-number/QBMRU7DWBGWVPD4LDWASC3N4AY/graph.json","events_json":"https://pith.science/api/pith-number/QBMRU7DWBGWVPD4LDWASC3N4AY/events.json","paper":"https://pith.science/paper/QBMRU7DW"},"agent_actions":{"view_html":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY","download_json":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY.json","view_paper":"https://pith.science/paper/QBMRU7DW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.12396&json=true","fetch_graph":"https://pith.science/api/pith-number/QBMRU7DWBGWVPD4LDWASC3N4AY/graph.json","fetch_events":"https://pith.science/api/pith-number/QBMRU7DWBGWVPD4LDWASC3N4AY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY/action/storage_attestation","attest_author":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY/action/author_attestation","sign_citation":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY/action/citation_signature","submit_replication":"https://pith.science/pith/QBMRU7DWBGWVPD4LDWASC3N4AY/action/replication_record"}},"created_at":"2026-07-05T07:58:00.789345+00:00","updated_at":"2026-07-05T07:58:00.789345+00:00"}