{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZN5BTRG5KOIEZGPVCEWCFFSFLH","short_pith_number":"pith:ZN5BTRG5","schema_version":"1.0","canonical_sha256":"cb7a19c4dd53904c99f5112c22964559c1f664def48121c40a58524996866535","source":{"kind":"arxiv","id":"2410.05650","version":1},"attestation_state":"computed","paper":{"title":"SIA-OVD: Shape-Invariant Adapter for Bridging the Image-Region Gap in Open-Vocabulary Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Jinglin Xu, Wenhao Zhou, Yuxin Peng, Zishuo Wang","submitted_at":"2024-10-08T02:59:08Z","abstract_excerpt":"Open-vocabulary detection (OVD) aims to detect novel objects without instance-level annotations to achieve open-world object detection at a lower cost. Existing OVD methods mainly rely on the powerful open-vocabulary image-text alignment capability of Vision-Language Pretrained Models (VLM) such as CLIP. However, CLIP is trained on image-text pairs and lacks the perceptual ability for local regions within an image, resulting in the gap between image and region representations. Directly using CLIP for OVD causes inaccurate region classification. We find the image-region gap is primarily caused "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05650","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-08T02:59:08Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"a42634dc77bf5fa5c2bc51d9be4b56bfa1b56dbb164aa199b4d3ebebe648b701","abstract_canon_sha256":"4e673c9bfdcf10b44cc2092edf9d35520fb39e1f47368164341e95d08098164e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:22.686833Z","signature_b64":"eedgVOlkYO4IRehgVCNRtvcB1cFZx//vViCgFIN8Nv/UbezteQx/sv92TugZtkPHpbcFY8kTa9OiC6Zk9yTiDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb7a19c4dd53904c99f5112c22964559c1f664def48121c40a58524996866535","last_reissued_at":"2026-07-05T09:17:22.686340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:22.686340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SIA-OVD: Shape-Invariant Adapter for Bridging the Image-Region Gap in Open-Vocabulary Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Jinglin Xu, Wenhao Zhou, Yuxin Peng, Zishuo Wang","submitted_at":"2024-10-08T02:59:08Z","abstract_excerpt":"Open-vocabulary detection (OVD) aims to detect novel objects without instance-level annotations to achieve open-world object detection at a lower cost. Existing OVD methods mainly rely on the powerful open-vocabulary image-text alignment capability of Vision-Language Pretrained Models (VLM) such as CLIP. However, CLIP is trained on image-text pairs and lacks the perceptual ability for local regions within an image, resulting in the gap between image and region representations. Directly using CLIP for OVD causes inaccurate region classification. We find the image-region gap is primarily caused "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05650","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05650","created_at":"2026-07-05T09:17:22.686403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05650v1","created_at":"2026-07-05T09:17:22.686403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05650","created_at":"2026-07-05T09:17:22.686403+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZN5BTRG5KOIE","created_at":"2026-07-05T09:17:22.686403+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZN5BTRG5KOIEZGPV","created_at":"2026-07-05T09:17:22.686403+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZN5BTRG5","created_at":"2026-07-05T09:17:22.686403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12120","citing_title":"To Whom Do Language Models Align? Measuring Principal Hierarchies Under High-Stakes Competing Demands","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH","json":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH.json","graph_json":"https://pith.science/api/pith-number/ZN5BTRG5KOIEZGPVCEWCFFSFLH/graph.json","events_json":"https://pith.science/api/pith-number/ZN5BTRG5KOIEZGPVCEWCFFSFLH/events.json","paper":"https://pith.science/paper/ZN5BTRG5"},"agent_actions":{"view_html":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH","download_json":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH.json","view_paper":"https://pith.science/paper/ZN5BTRG5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05650&json=true","fetch_graph":"https://pith.science/api/pith-number/ZN5BTRG5KOIEZGPVCEWCFFSFLH/graph.json","fetch_events":"https://pith.science/api/pith-number/ZN5BTRG5KOIEZGPVCEWCFFSFLH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH/action/storage_attestation","attest_author":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH/action/author_attestation","sign_citation":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH/action/citation_signature","submit_replication":"https://pith.science/pith/ZN5BTRG5KOIEZGPVCEWCFFSFLH/action/replication_record"}},"created_at":"2026-07-05T09:17:22.686403+00:00","updated_at":"2026-07-05T09:17:22.686403+00:00"}