{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EVLFUC5UNGG7JH7OAH7CUXMJMY","short_pith_number":"pith:EVLFUC5U","schema_version":"1.0","canonical_sha256":"25565a0bb4698df49fee01fe2a5d8966076907ab64387f4fdf75a12689a9f29a","source":{"kind":"arxiv","id":"2405.10300","version":2},"attestation_state":"computed","paper":{"title":"Grounding DINO 1.5: Advance the \"Edge\" of Open-Set Object Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Li, Han Gao, Hao Zhang, Hongjie Huang, Kent Yu, Lei Zhang, Peijun Tang, Qing Jiang, Shilong Liu, Tianhe Ren, Wenlong Liu, Xiaoke Jiang, Yihao Chen, Yuda Xiong, Zhaoyang Zeng, Zhengyu Ma","submitted_at":"2024-05-16T17:54:15Z","abstract_excerpt":"This paper introduces Grounding DINO 1.5, a suite of advanced open-set object detection models developed by IDEA Research, which aims to advance the \"Edge\" of open-set object detection. The suite encompasses two models: Grounding DINO 1.5 Pro, a high-performance model designed for stronger generalization capability across a wide range of scenarios, and Grounding DINO 1.5 Edge, an efficient model optimized for faster speed demanded in many applications requiring edge deployment. The Grounding DINO 1.5 Pro model advances its predecessor by scaling up the model architecture, integrating an enhanc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.10300","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-16T17:54:15Z","cross_cats_sorted":[],"title_canon_sha256":"1d4fed1deb8a8317ea3626f2e5fed21b752d91cca3d42dd55065f5fb13359662","abstract_canon_sha256":"9894819fd2a4f047f4cde7adf7971e500f1c2eb169bdb810d38b635f2cdc8e78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:54.104060Z","signature_b64":"YmTA8mOiY4ffB16VUW3dP1TJJKf7LEbVKX55fC4atgJT5XpRW2s1OPQj+mZMznG8wP8U4stPbtdFaquJXStlAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25565a0bb4698df49fee01fe2a5d8966076907ab64387f4fdf75a12689a9f29a","last_reissued_at":"2026-07-05T08:25:54.103576Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:54.103576Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grounding DINO 1.5: Advance the \"Edge\" of Open-Set Object Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feng Li, Han Gao, Hao Zhang, Hongjie Huang, Kent Yu, Lei Zhang, Peijun Tang, Qing Jiang, Shilong Liu, Tianhe Ren, Wenlong Liu, Xiaoke Jiang, Yihao Chen, Yuda Xiong, Zhaoyang Zeng, Zhengyu Ma","submitted_at":"2024-05-16T17:54:15Z","abstract_excerpt":"This paper introduces Grounding DINO 1.5, a suite of advanced open-set object detection models developed by IDEA Research, which aims to advance the \"Edge\" of open-set object detection. The suite encompasses two models: Grounding DINO 1.5 Pro, a high-performance model designed for stronger generalization capability across a wide range of scenarios, and Grounding DINO 1.5 Edge, an efficient model optimized for faster speed demanded in many applications requiring edge deployment. The Grounding DINO 1.5 Pro model advances its predecessor by scaling up the model architecture, integrating an enhanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.10300","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.10300/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.10300","created_at":"2026-07-05T08:25:54.103635+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.10300v2","created_at":"2026-07-05T08:25:54.103635+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.10300","created_at":"2026-07-05T08:25:54.103635+00:00"},{"alias_kind":"pith_short_12","alias_value":"EVLFUC5UNGG7","created_at":"2026-07-05T08:25:54.103635+00:00"},{"alias_kind":"pith_short_16","alias_value":"EVLFUC5UNGG7JH7O","created_at":"2026-07-05T08:25:54.103635+00:00"},{"alias_kind":"pith_short_8","alias_value":"EVLFUC5U","created_at":"2026-07-05T08:25:54.103635+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27036","citing_title":"RelAfford6D: Relational 6D Affordance Graphs for Constraint-Driven Robotic Manipulation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23653","citing_title":"Lightweight Neural Framework for Robust 3D Volume and Surface Estimation from Multi-View Images","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11783","citing_title":"A Comprehensive Ecosystem for Open-Domain Customized Video Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11546","citing_title":"VL-DINO: Leveraging CLIP Vision-Language Knowledge for Open-Vocabulary Object Detectio","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09882","citing_title":"WHU-Infra3D: A Full-stack Multi-modal Dataset and Benchmark for 3D Roadside Infrastructure Inventory","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03748","citing_title":"Ultralytics YOLO26: Unified Real-Time End-to-End Vision Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14923","citing_title":"SceneParser: Hierarchical Scene Parsing for Visual Semantics Understanding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26576","citing_title":"TrackRef3D: Multi-View Consistent Track-then-Label for Open-World Referring Segmentation in 3D Gaussian Splatting","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00782","citing_title":"FlowOVD: Learning Generative Latent Flows for Zero-shot Open-vocabulary Detection","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2501.08083","citing_title":"Benchmarking Vision Foundation Models for Input Monitoring in Autonomous Driving","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13923","citing_title":"Qwen2.5-VL Technical Report","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23622","citing_title":"DLEBench: Evaluating Small-scale Object Editing Ability for Instruction-based Image Editing Model","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19410","citing_title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17014","citing_title":"RHINO: Reconstructing Human Interactions with Novel Objects from Monocular Videos","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02311","citing_title":"Inferring Dynamic Physical Properties from Video Foundation Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17001","citing_title":"Unify Robot Actions in Camera Frame","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16719","citing_title":"SAM 3: Segment Anything with Concepts","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01925","citing_title":"A Survey on Vision-Language-Action Models: An Action Tokenization Perspective","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2504.07615","citing_title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10190","citing_title":"DetRefiner: Model-Agnostic Detection Refinement with Feature Fusion Transformer","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11789","citing_title":"LMMs Meet Object-Centric Vision: Understanding, Segmentation, Editing and Generation","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05180","citing_title":"MIRAGE: Benchmarking and Aligning Multi-Instance Image Editing","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY","json":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY.json","graph_json":"https://pith.science/api/pith-number/EVLFUC5UNGG7JH7OAH7CUXMJMY/graph.json","events_json":"https://pith.science/api/pith-number/EVLFUC5UNGG7JH7OAH7CUXMJMY/events.json","paper":"https://pith.science/paper/EVLFUC5U"},"agent_actions":{"view_html":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY","download_json":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY.json","view_paper":"https://pith.science/paper/EVLFUC5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.10300&json=true","fetch_graph":"https://pith.science/api/pith-number/EVLFUC5UNGG7JH7OAH7CUXMJMY/graph.json","fetch_events":"https://pith.science/api/pith-number/EVLFUC5UNGG7JH7OAH7CUXMJMY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY/action/storage_attestation","attest_author":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY/action/author_attestation","sign_citation":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY/action/citation_signature","submit_replication":"https://pith.science/pith/EVLFUC5UNGG7JH7OAH7CUXMJMY/action/replication_record"}},"created_at":"2026-07-05T08:25:54.103635+00:00","updated_at":"2026-07-05T08:25:54.103635+00:00"}