{"work":{"id":"876f9fe8-c712-4550-9c26-cc18ab69abf2","openalex_id":"https://openalex.org/W3092462694","doi":"10.48550/arxiv.2010.04159","arxiv_id":"2010.04159","raw_key":null,"title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","authors":null,"authors_text":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, Jifeng Dai","year":2020,"venue":"cs.CV","abstract":"DETR has been recently proposed to eliminate the need for many hand-designed components in object detection while demonstrating good performance. However, it suffers from slow convergence and limited feature spatial resolution, due to the limitation of Transformer attention modules in processing image feature maps. To mitigate these issues, we proposed Deformable DETR, whose attention modules only attend to a small set of key sampling points around a reference. Deformable DETR can achieve better performance than DETR (especially on small objects) with 10 times less training epochs. Extensive experiments on the COCO benchmark demonstrate the effectiveness of our approach. Code is released at https://github.com/fundamentalvision/Deformable-DETR.","external_url":"https://arxiv.org/abs/2010.04159","cited_by_count":1868,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2010.04159","created_at":"2026-05-10T05:41:02.459416+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","render_title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection"},"hub":{"state":{"work_id":"876f9fe8-c712-4550-9c26-cc18ab69abf2","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":98,"external_cited_by_count":1868,"distinct_field_count":8,"first_pith_cited_at":"2024-03-25T08:57:27+00:00","last_pith_cited_at":"2026-07-09T00:11:48+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-23T06:19:29.435335+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":7},{"context_role":"method","n":2},{"context_role":"baseline","n":1}],"polarity_counts":[{"context_polarity":"background","n":7},{"context_polarity":"use_method","n":2},{"context_polarity":"baseline","n":1}],"runs":{"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-14T16:22:35.863429+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":9},{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":8},{"title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","work_id":"4e842f69-fa99-4dde-b8a7-b5a558d4c80b","shared_citers":6},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":5},{"title":"YOLOv4: Optimal Speed and Accuracy of Object Detection","work_id":"7057aaee-27f6-4209-a83c-f59727f937a8","shared_citers":5},{"title":"YOLOX: Exceeding YOLO Series in 2021","work_id":"112b3cd9-8fe6-49fe-bbaa-90a3f46045c7","shared_citers":5},{"title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","shared_citers":4},{"title":"SAM 2: Segment Anything in Images and Videos","work_id":"acc13f66-d814-44f9-9688-375688bf2d4a","shared_citers":4},{"title":"BEVDet: High-performance multi-camera 3D object detection in bird-eye-view","work_id":"cbbff760-e4d5-44da-bbe3-8286f8bf1ad6","shared_citers":3},{"title":"DAB-DETR: Dynamic anchor boxes are better queries for DETR","work_id":"2e01e759-de8e-43c2-9c99-f49e3edcb179","shared_citers":3},{"title":"End-to-end object detection with transformers","work_id":"f65d79c7-1ade-4a0e-b71e-140e377dbfb4","shared_citers":3},{"title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","shared_citers":3},{"title":"Ob- jects as points","work_id":"3567080d-f164-46ab-903a-02853db3a970","shared_citers":3},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":3},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":3},{"title":"Rt-detrv2: Improved base- line with bag-of-freebies for real-time detection transformer","work_id":"1a8c480b-eab7-46c0-815b-8ffabeb6d9ef","shared_citers":3},{"title":"Vision Mamba: Efficient Visual Representation Learning with Bidirectional State Space Model","work_id":"bd81352e-a64f-4720-9f76-ddda0ea9af83","shared_citers":3},{"title":"YOLOv11: An Overview of the Key Architectural Enhancements","work_id":"17e84c13-a2a4-4b25-9b05-ca4ca212abfa","shared_citers":3},{"title":"YOLOv3: An Incremental Improvement","work_id":"d737b3cc-9bd1-43d6-8310-c91e64b510f7","shared_citers":3},{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":2},{"title":"Argoverse 2: Next Generation Datasets for Self-Driving Perception and Forecasting","work_id":"f2f765af-f0ba-41f1-8dd1-37bf1d134d31","shared_citers":2},{"title":"arXiv preprint arXiv:2003.09003 (2020)","work_id":"547de000-c99d-4567-a1f2-110b22feca91","shared_citers":2},{"title":"arXiv preprint arXiv:2012.15460 (2020) 19","work_id":"6663b6f7-c745-4dad-93e5-8b1ee2adb970","shared_citers":2},{"title":"arXiv preprint arXiv:2206.14651 (2022)","work_id":"450934b5-2ed0-4d79-9dc2-cf3d20caaf84","shared_citers":2}],"time_series":[{"n":1,"year":2025},{"n":43,"year":2026}],"dependency_candidates":[]},"error":null,"updated_at":"2026-05-14T16:22:20.972075+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"items":[{"title":"Qwen3 Technical Report","outcome":"unchanged","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"counts":{"fixed":0,"merged":0,"unchanged":1,"quarantined":0,"needs_external_resolution":0},"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-14T16:22:26.077010+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","claims":[{"claim_text":"DETR has been recently proposed to eliminate the need for many hand-designed components in object detection while demonstrating good performance. However, it suffers from slow convergence and limited feature spatial resolution, due to the limitation of Transformer attention modules in processing image feature maps. To mitigate these issues, we proposed Deformable DETR, whose attention modules only attend to a small set of key sampling points around a reference. Deformable DETR can achieve better performance than DETR (especially on small objects) with 10 times less training epochs. Extensive e","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Deformable DETR: Deformable Transformers for End-to-End Object Detection because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T16:22:30.713562+00:00"}},"summary":{"title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","claims":[{"claim_text":"DETR has been recently proposed to eliminate the need for many hand-designed components in object detection while demonstrating good performance. However, it suffers from slow convergence and limited feature spatial resolution, due to the limitation of Transformer attention modules in processing image feature maps. To mitigate these issues, we proposed Deformable DETR, whose attention modules only attend to a small set of key sampling points around a reference. Deformable DETR can achieve better performance than DETR (especially on small objects) with 10 times less training epochs. Extensive e","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Deformable DETR: Deformable Transformers for End-to-End Object Detection because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":9},{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":8},{"title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","work_id":"4e842f69-fa99-4dde-b8a7-b5a558d4c80b","shared_citers":6},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":5},{"title":"YOLOv4: Optimal Speed and Accuracy of Object Detection","work_id":"7057aaee-27f6-4209-a83c-f59727f937a8","shared_citers":5},{"title":"YOLOX: Exceeding YOLO Series in 2021","work_id":"112b3cd9-8fe6-49fe-bbaa-90a3f46045c7","shared_citers":5},{"title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","shared_citers":4},{"title":"SAM 2: Segment Anything in Images and Videos","work_id":"acc13f66-d814-44f9-9688-375688bf2d4a","shared_citers":4},{"title":"BEVDet: High-performance multi-camera 3D object detection in bird-eye-view","work_id":"cbbff760-e4d5-44da-bbe3-8286f8bf1ad6","shared_citers":3},{"title":"DAB-DETR: Dynamic anchor boxes are better queries for DETR","work_id":"2e01e759-de8e-43c2-9c99-f49e3edcb179","shared_citers":3},{"title":"End-to-end object detection with transformers","work_id":"f65d79c7-1ade-4a0e-b71e-140e377dbfb4","shared_citers":3},{"title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","shared_citers":3},{"title":"Ob- jects as points","work_id":"3567080d-f164-46ab-903a-02853db3a970","shared_citers":3},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":3},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":3},{"title":"Rt-detrv2: Improved base- line with bag-of-freebies for real-time detection transformer","work_id":"1a8c480b-eab7-46c0-815b-8ffabeb6d9ef","shared_citers":3},{"title":"Vision Mamba: Efficient Visual Representation Learning with Bidirectional State Space Model","work_id":"bd81352e-a64f-4720-9f76-ddda0ea9af83","shared_citers":3},{"title":"YOLOv11: An Overview of the Key Architectural Enhancements","work_id":"17e84c13-a2a4-4b25-9b05-ca4ca212abfa","shared_citers":3},{"title":"YOLOv3: An Incremental Improvement","work_id":"d737b3cc-9bd1-43d6-8310-c91e64b510f7","shared_citers":3},{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":2},{"title":"Argoverse 2: Next Generation Datasets for Self-Driving Perception and Forecasting","work_id":"f2f765af-f0ba-41f1-8dd1-37bf1d134d31","shared_citers":2},{"title":"arXiv preprint arXiv:2003.09003 (2020)","work_id":"547de000-c99d-4567-a1f2-110b22feca91","shared_citers":2},{"title":"arXiv preprint arXiv:2012.15460 (2020) 19","work_id":"6663b6f7-c745-4dad-93e5-8b1ee2adb970","shared_citers":2},{"title":"arXiv preprint arXiv:2206.14651 (2022)","work_id":"450934b5-2ed0-4d79-9dc2-cf3d20caaf84","shared_citers":2}],"time_series":[{"n":1,"year":2025},{"n":43,"year":2026}],"dependency_candidates":[]},"authors":[]}}