{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JSC5ZZF3IAJK7RWSJ5QX2SGT5U","short_pith_number":"pith:JSC5ZZF3","schema_version":"1.0","canonical_sha256":"4c85dce4bb4012afc6d24f617d48d3ed073b7d933a365aa04aaa62605e9077dc","source":{"kind":"arxiv","id":"2209.13959","version":2},"attestation_state":"computed","paper":{"title":"Dynamic MDETR: A Dynamic Multimodal Transformer Decoder for Visual Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fengyuan Shi, Limin Wang, Ruopeng Gao, Weilin Huang","submitted_at":"2022-09-28T09:43:02Z","abstract_excerpt":"Multimodal transformer exhibits high capacity and flexibility to align image and text for visual grounding. However, the existing encoder-only grounding framework (e.g., TransVG) suffers from heavy computation due to the self-attention operation with quadratic time complexity. To address this issue, we present a new multimodal transformer architecture, coined as Dynamic Mutilmodal DETR (Dynamic MDETR), by decoupling the whole grounding process into encoding and decoding phases. The key observation is that there exists high spatial redundancy in images. Thus, we devise a new dynamic multimodal "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.13959","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-09-28T09:43:02Z","cross_cats_sorted":[],"title_canon_sha256":"6c4b9cb0b0e66e01108852e0e9baa64e0cb9f0babadd29108c1a3c45c3910827","abstract_canon_sha256":"1a0157c0088c56eac9e0fa8719074ec47ddd7b6fe90089638d46f9ba0d1c8937"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:05.203632Z","signature_b64":"dOasFXTbBZ7RchlSsPl1praXrM+00hcc4aXdBsEJMy1mZGcnFdywbh80jdCoD/zv9yJLdjtDTAM1lNafDQU5Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c85dce4bb4012afc6d24f617d48d3ed073b7d933a365aa04aaa62605e9077dc","last_reissued_at":"2026-07-05T07:05:05.203110Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:05.203110Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamic MDETR: A Dynamic Multimodal Transformer Decoder for Visual Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fengyuan Shi, Limin Wang, Ruopeng Gao, Weilin Huang","submitted_at":"2022-09-28T09:43:02Z","abstract_excerpt":"Multimodal transformer exhibits high capacity and flexibility to align image and text for visual grounding. However, the existing encoder-only grounding framework (e.g., TransVG) suffers from heavy computation due to the self-attention operation with quadratic time complexity. To address this issue, we present a new multimodal transformer architecture, coined as Dynamic Mutilmodal DETR (Dynamic MDETR), by decoupling the whole grounding process into encoding and decoding phases. The key observation is that there exists high spatial redundancy in images. Thus, we devise a new dynamic multimodal "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.13959","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.13959/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.13959","created_at":"2026-07-05T07:05:05.203175+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.13959v2","created_at":"2026-07-05T07:05:05.203175+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.13959","created_at":"2026-07-05T07:05:05.203175+00:00"},{"alias_kind":"pith_short_12","alias_value":"JSC5ZZF3IAJK","created_at":"2026-07-05T07:05:05.203175+00:00"},{"alias_kind":"pith_short_16","alias_value":"JSC5ZZF3IAJK7RWS","created_at":"2026-07-05T07:05:05.203175+00:00"},{"alias_kind":"pith_short_8","alias_value":"JSC5ZZF3","created_at":"2026-07-05T07:05:05.203175+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U","json":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U.json","graph_json":"https://pith.science/api/pith-number/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/graph.json","events_json":"https://pith.science/api/pith-number/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/events.json","paper":"https://pith.science/paper/JSC5ZZF3"},"agent_actions":{"view_html":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U","download_json":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U.json","view_paper":"https://pith.science/paper/JSC5ZZF3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.13959&json=true","fetch_graph":"https://pith.science/api/pith-number/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/graph.json","fetch_events":"https://pith.science/api/pith-number/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/action/storage_attestation","attest_author":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/action/author_attestation","sign_citation":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/action/citation_signature","submit_replication":"https://pith.science/pith/JSC5ZZF3IAJK7RWSJ5QX2SGT5U/action/replication_record"}},"created_at":"2026-07-05T07:05:05.203175+00:00","updated_at":"2026-07-05T07:05:05.203175+00:00"}