{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:KUZVR6YPOAOWSFOBIGD5WCJ5KZ","short_pith_number":"pith:KUZVR6YP","schema_version":"1.0","canonical_sha256":"553358fb0f701d6915c14187db093d566580b665cd312875f509b8f0695629ea","source":{"kind":"arxiv","id":"2004.13278","version":3},"attestation_state":"computed","paper":{"title":"VD-BERT: A Unified Vision and Dialog Transformer with BERT","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Caiming Xiong, Irwin King, Michael R. Lyu, Shafiq Joty, Steven C.H. Hoi, Yue Wang","submitted_at":"2020-04-28T04:08:46Z","abstract_excerpt":"Visual dialog is a challenging vision-language task, where a dialog agent needs to answer a series of questions through reasoning on the image content and dialog history. Prior work has mostly focused on various attention mechanisms to model such intricate interactions. By contrast, in this work, we propose VD-BERT, a simple yet effective framework of unified vision-dialog Transformer that leverages the pretrained BERT language models for Visual Dialog tasks. The model is unified in that (1) it captures all the interactions between the image and the multi-turn dialog using a single-stream Tran"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.13278","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2020-04-28T04:08:46Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2008089efbe0a85ff86339df797eb71dcea52738f351fd76a7f77200760ad393","abstract_canon_sha256":"8c7865588d5e3701a8cc60064d09c72dea45018bc707c448e39b7a71be61acd5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:48:09.161858Z","signature_b64":"CI0NZH2z9ymk98iDc1LU6xvuaItZvBv8hCTkFUMzVQ7gQ8pxUjvBN4+VwCZ/M0C9/DbueaynjmsSHgS90aq0DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"553358fb0f701d6915c14187db093d566580b665cd312875f509b8f0695629ea","last_reissued_at":"2026-07-05T01:48:09.161451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:48:09.161451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VD-BERT: A Unified Vision and Dialog Transformer with BERT","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Caiming Xiong, Irwin King, Michael R. Lyu, Shafiq Joty, Steven C.H. Hoi, Yue Wang","submitted_at":"2020-04-28T04:08:46Z","abstract_excerpt":"Visual dialog is a challenging vision-language task, where a dialog agent needs to answer a series of questions through reasoning on the image content and dialog history. Prior work has mostly focused on various attention mechanisms to model such intricate interactions. By contrast, in this work, we propose VD-BERT, a simple yet effective framework of unified vision-dialog Transformer that leverages the pretrained BERT language models for Visual Dialog tasks. The model is unified in that (1) it captures all the interactions between the image and the multi-turn dialog using a single-stream Tran"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.13278","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.13278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.13278","created_at":"2026-07-05T01:48:09.161508+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.13278v3","created_at":"2026-07-05T01:48:09.161508+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.13278","created_at":"2026-07-05T01:48:09.161508+00:00"},{"alias_kind":"pith_short_12","alias_value":"KUZVR6YPOAOW","created_at":"2026-07-05T01:48:09.161508+00:00"},{"alias_kind":"pith_short_16","alias_value":"KUZVR6YPOAOWSFOB","created_at":"2026-07-05T01:48:09.161508+00:00"},{"alias_kind":"pith_short_8","alias_value":"KUZVR6YP","created_at":"2026-07-05T01:48:09.161508+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24020","citing_title":"Machine Intelligence that Understands Visual and Linguistic Information and Interacts with Humans and Environments","ref_index":123,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ","json":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ.json","graph_json":"https://pith.science/api/pith-number/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/graph.json","events_json":"https://pith.science/api/pith-number/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/events.json","paper":"https://pith.science/paper/KUZVR6YP"},"agent_actions":{"view_html":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ","download_json":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ.json","view_paper":"https://pith.science/paper/KUZVR6YP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.13278&json=true","fetch_graph":"https://pith.science/api/pith-number/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/graph.json","fetch_events":"https://pith.science/api/pith-number/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/action/storage_attestation","attest_author":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/action/author_attestation","sign_citation":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/action/citation_signature","submit_replication":"https://pith.science/pith/KUZVR6YPOAOWSFOBIGD5WCJ5KZ/action/replication_record"}},"created_at":"2026-07-05T01:48:09.161508+00:00","updated_at":"2026-07-05T01:48:09.161508+00:00"}