{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:CIHPMTKJOHCSLS4YWQ5COCBGUK","short_pith_number":"pith:CIHPMTKJ","schema_version":"1.0","canonical_sha256":"120ef64d4971c525cb98b43a270826a28a31f4091a7d7b464831bf110489a64e","source":{"kind":"arxiv","id":"2104.01318","version":1},"attestation_state":"computed","paper":{"title":"Efficient DETR: Improving End-to-End Object Detector with Dense Prior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boxun Li, Chi Zhang, Jiangbo Ai, Zhuyu Yao","submitted_at":"2021-04-03T06:14:24Z","abstract_excerpt":"The recently proposed end-to-end transformer detectors, such as DETR and Deformable DETR, have a cascade structure of stacking 6 decoder layers to update object queries iteratively, without which their performance degrades seriously. In this paper, we investigate that the random initialization of object containers, which include object queries and reference points, is mainly responsible for the requirement of multiple iterations. Based on our findings, we propose Efficient DETR, a simple and efficient pipeline for end-to-end object detection. By taking advantage of both dense detection and spa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.01318","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-04-03T06:14:24Z","cross_cats_sorted":[],"title_canon_sha256":"b07a444cba815a22e6ee261854ce8600f4797303d115e79fb3c2064249a2f3d3","abstract_canon_sha256":"f9b5cfa6c6bfa6b88d16ca6633d7092f754cadaa418657fcde1e24bc9fa8421c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:28:58.918778Z","signature_b64":"C7L8TJoiSdURwiW6Zx9ZX7m/0ATdBs/R6kiD5tI9XGGnJeQ33IEzx2VQo6FkkZkop8wR/T2K8sVJ/CyHOPlPBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"120ef64d4971c525cb98b43a270826a28a31f4091a7d7b464831bf110489a64e","last_reissued_at":"2026-07-05T02:28:58.918342Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:28:58.918342Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient DETR: Improving End-to-End Object Detector with Dense Prior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Boxun Li, Chi Zhang, Jiangbo Ai, Zhuyu Yao","submitted_at":"2021-04-03T06:14:24Z","abstract_excerpt":"The recently proposed end-to-end transformer detectors, such as DETR and Deformable DETR, have a cascade structure of stacking 6 decoder layers to update object queries iteratively, without which their performance degrades seriously. In this paper, we investigate that the random initialization of object containers, which include object queries and reference points, is mainly responsible for the requirement of multiple iterations. Based on our findings, we propose Efficient DETR, a simple and efficient pipeline for end-to-end object detection. By taking advantage of both dense detection and spa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.01318","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.01318/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.01318","created_at":"2026-07-05T02:28:58.918401+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.01318v1","created_at":"2026-07-05T02:28:58.918401+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.01318","created_at":"2026-07-05T02:28:58.918401+00:00"},{"alias_kind":"pith_short_12","alias_value":"CIHPMTKJOHCS","created_at":"2026-07-05T02:28:58.918401+00:00"},{"alias_kind":"pith_short_16","alias_value":"CIHPMTKJOHCSLS4Y","created_at":"2026-07-05T02:28:58.918401+00:00"},{"alias_kind":"pith_short_8","alias_value":"CIHPMTKJ","created_at":"2026-07-05T02:28:58.918401+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22702","citing_title":"Modular Diffusion Models for Structured Visual Recognition","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00124","citing_title":"Segmenting, Fast and Slow: Real-Time Open-Vocabulary Video Instance Segmentation with Dual-Path Processing","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27831","citing_title":"Hippocampus-DETR: An Explicit Memory Object Detection Framework Based on Hippocampus Modeling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14110","citing_title":"SToRe3D: Sparse Token Relevance in ViTs for Efficient Multi-View 3D Object Detection","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2203.03605","citing_title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK","json":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK.json","graph_json":"https://pith.science/api/pith-number/CIHPMTKJOHCSLS4YWQ5COCBGUK/graph.json","events_json":"https://pith.science/api/pith-number/CIHPMTKJOHCSLS4YWQ5COCBGUK/events.json","paper":"https://pith.science/paper/CIHPMTKJ"},"agent_actions":{"view_html":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK","download_json":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK.json","view_paper":"https://pith.science/paper/CIHPMTKJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.01318&json=true","fetch_graph":"https://pith.science/api/pith-number/CIHPMTKJOHCSLS4YWQ5COCBGUK/graph.json","fetch_events":"https://pith.science/api/pith-number/CIHPMTKJOHCSLS4YWQ5COCBGUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK/action/storage_attestation","attest_author":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK/action/author_attestation","sign_citation":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK/action/citation_signature","submit_replication":"https://pith.science/pith/CIHPMTKJOHCSLS4YWQ5COCBGUK/action/replication_record"}},"created_at":"2026-07-05T02:28:58.918401+00:00","updated_at":"2026-07-05T02:28:58.918401+00:00"}