{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:NKNEIJNF2G7SH3GMMUNNHPPMWJ","short_pith_number":"pith:NKNEIJNF","schema_version":"1.0","canonical_sha256":"6a9a4425a5d1bf23eccc651ad3bdecb253eac2bd09436cf1fec682775b2d56a9","source":{"kind":"arxiv","id":"2104.12763","version":2},"attestation_state":"computed","paper":{"title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aishwarya Kamath, Gabriel Synnaeve, Ishan Misra, Mannat Singh, Nicolas Carion, Yann LeCun","submitted_at":"2021-04-26T17:55:33Z","abstract_excerpt":"Multi-modal reasoning systems rely on a pre-trained object detector to extract regions of interest from the image. However, this crucial module is typically used as a black box, trained independently of the downstream task and on a fixed vocabulary of objects and attributes. This makes it challenging for such systems to capture the long tail of visual concepts expressed in free form text. In this paper we propose MDETR, an end-to-end modulated detector that detects objects in an image conditioned on a raw text query, like a caption or a question. We use a transformer-based architecture to reas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.12763","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-04-26T17:55:33Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"f445f75a4cd57b4390d7c15d4bb2b56ac2e4fb0ab7d389282df22d77b9327d51","abstract_canon_sha256":"239294aac396706f7b92aa0254f7a95955db7ff54a128dbd894537cda1f2ff24"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:21:47.541912Z","signature_b64":"W0Yf49TSGJFy0fvr6QYH0yHvOYKOTOoDHQLPrT1SUdih/fyzePnACFoJ0QnOTassdi0NjDtDCHkXbCpCNXEqAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a9a4425a5d1bf23eccc651ad3bdecb253eac2bd09436cf1fec682775b2d56a9","last_reissued_at":"2026-07-05T03:21:47.541412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:21:47.541412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aishwarya Kamath, Gabriel Synnaeve, Ishan Misra, Mannat Singh, Nicolas Carion, Yann LeCun","submitted_at":"2021-04-26T17:55:33Z","abstract_excerpt":"Multi-modal reasoning systems rely on a pre-trained object detector to extract regions of interest from the image. However, this crucial module is typically used as a black box, trained independently of the downstream task and on a fixed vocabulary of objects and attributes. This makes it challenging for such systems to capture the long tail of visual concepts expressed in free form text. In this paper we propose MDETR, an end-to-end modulated detector that detects objects in an image conditioned on a raw text query, like a caption or a question. We use a transformer-based architecture to reas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.12763","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.12763/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.12763","created_at":"2026-07-05T03:21:47.541474+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.12763v2","created_at":"2026-07-05T03:21:47.541474+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.12763","created_at":"2026-07-05T03:21:47.541474+00:00"},{"alias_kind":"pith_short_12","alias_value":"NKNEIJNF2G7S","created_at":"2026-07-05T03:21:47.541474+00:00"},{"alias_kind":"pith_short_16","alias_value":"NKNEIJNF2G7SH3GM","created_at":"2026-07-05T03:21:47.541474+00:00"},{"alias_kind":"pith_short_8","alias_value":"NKNEIJNF","created_at":"2026-07-05T03:21:47.541474+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2208.14649","citing_title":"DetailCLIP: Injecting Image Details into CLIP's Feature Space","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19410","citing_title":"Vision Harnessing Agent for Open Ad-hoc Segmentation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18434","citing_title":"TIGER-FG: Text-Guided Implicit Fine-Grained Grounding for E-commerce Retrieval","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10527","citing_title":"STORM: End-to-End Referring Multi-Object Tracking in Videos","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17488","citing_title":"AutoVQA-G: Self-Improving Agentic Framework for Automated Visual Question Answering and Grounding Annotation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ","json":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ.json","graph_json":"https://pith.science/api/pith-number/NKNEIJNF2G7SH3GMMUNNHPPMWJ/graph.json","events_json":"https://pith.science/api/pith-number/NKNEIJNF2G7SH3GMMUNNHPPMWJ/events.json","paper":"https://pith.science/paper/NKNEIJNF"},"agent_actions":{"view_html":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ","download_json":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ.json","view_paper":"https://pith.science/paper/NKNEIJNF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.12763&json=true","fetch_graph":"https://pith.science/api/pith-number/NKNEIJNF2G7SH3GMMUNNHPPMWJ/graph.json","fetch_events":"https://pith.science/api/pith-number/NKNEIJNF2G7SH3GMMUNNHPPMWJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ/action/storage_attestation","attest_author":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ/action/author_attestation","sign_citation":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ/action/citation_signature","submit_replication":"https://pith.science/pith/NKNEIJNF2G7SH3GMMUNNHPPMWJ/action/replication_record"}},"created_at":"2026-07-05T03:21:47.541474+00:00","updated_at":"2026-07-05T03:21:47.541474+00:00"}