{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GX4LSJZOEU5MUNSA5DKCHHG3XH","short_pith_number":"pith:GX4LSJZO","schema_version":"1.0","canonical_sha256":"35f8b9272e253aca3640e8d4239cdbb9d05aa6ecce1c9e8b47d71d8d6a2af845","source":{"kind":"arxiv","id":"2109.08478","version":1},"attestation_state":"computed","paper":{"title":"Multimodal Incremental Transformer with Visual Grounding for Visual Dialogue Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.MM"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Feilong Chen, Jie Zhou, Peng Li, Xiuyi Chen","submitted_at":"2021-09-17T11:39:29Z","abstract_excerpt":"Visual dialogue is a challenging task since it needs to answer a series of coherent questions on the basis of understanding the visual environment. Previous studies focus on the implicit exploration of multimodal co-reference by implicitly attending to spatial image features or object-level image features but neglect the importance of locating the objects explicitly in the visual content, which is associated with entities in the textual content. Therefore, in this paper we propose a {\\bf M}ultimodal {\\bf I}ncremental {\\bf T}ransformer with {\\bf V}isual {\\bf G}rounding, named MITVG, which consi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.08478","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-09-17T11:39:29Z","cross_cats_sorted":["cs.CV","cs.MM"],"title_canon_sha256":"cea07785c9bd54574512bcd0eb8e1e798d31044ceff592f2ef0bf1e31149cd22","abstract_canon_sha256":"f41a45bb3d8643d18beb13f2c5ce772d239b73a70679559d08489f1fd05df4f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:15:16.359265Z","signature_b64":"PX6uNey0eGc6rYZIdR7DXWC+0f+kYr/d9KglMUTI+a1CuZ//uMs9f0wlEnSke2RpNNvav0rWw+O1pf9D0D2fDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35f8b9272e253aca3640e8d4239cdbb9d05aa6ecce1c9e8b47d71d8d6a2af845","last_reissued_at":"2026-07-05T03:15:16.358744Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:15:16.358744Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Incremental Transformer with Visual Grounding for Visual Dialogue Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.MM"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Feilong Chen, Jie Zhou, Peng Li, Xiuyi Chen","submitted_at":"2021-09-17T11:39:29Z","abstract_excerpt":"Visual dialogue is a challenging task since it needs to answer a series of coherent questions on the basis of understanding the visual environment. Previous studies focus on the implicit exploration of multimodal co-reference by implicitly attending to spatial image features or object-level image features but neglect the importance of locating the objects explicitly in the visual content, which is associated with entities in the textual content. Therefore, in this paper we propose a {\\bf M}ultimodal {\\bf I}ncremental {\\bf T}ransformer with {\\bf V}isual {\\bf G}rounding, named MITVG, which consi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.08478","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.08478/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.08478","created_at":"2026-07-05T03:15:16.358819+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.08478v1","created_at":"2026-07-05T03:15:16.358819+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.08478","created_at":"2026-07-05T03:15:16.358819+00:00"},{"alias_kind":"pith_short_12","alias_value":"GX4LSJZOEU5M","created_at":"2026-07-05T03:15:16.358819+00:00"},{"alias_kind":"pith_short_16","alias_value":"GX4LSJZOEU5MUNSA","created_at":"2026-07-05T03:15:16.358819+00:00"},{"alias_kind":"pith_short_8","alias_value":"GX4LSJZO","created_at":"2026-07-05T03:15:16.358819+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.12739","citing_title":"Transformer-based Spatial Grounding: A Comprehensive Survey","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH","json":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH.json","graph_json":"https://pith.science/api/pith-number/GX4LSJZOEU5MUNSA5DKCHHG3XH/graph.json","events_json":"https://pith.science/api/pith-number/GX4LSJZOEU5MUNSA5DKCHHG3XH/events.json","paper":"https://pith.science/paper/GX4LSJZO"},"agent_actions":{"view_html":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH","download_json":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH.json","view_paper":"https://pith.science/paper/GX4LSJZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.08478&json=true","fetch_graph":"https://pith.science/api/pith-number/GX4LSJZOEU5MUNSA5DKCHHG3XH/graph.json","fetch_events":"https://pith.science/api/pith-number/GX4LSJZOEU5MUNSA5DKCHHG3XH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH/action/storage_attestation","attest_author":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH/action/author_attestation","sign_citation":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH/action/citation_signature","submit_replication":"https://pith.science/pith/GX4LSJZOEU5MUNSA5DKCHHG3XH/action/replication_record"}},"created_at":"2026-07-05T03:15:16.358819+00:00","updated_at":"2026-07-05T03:15:16.358819+00:00"}