{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:ZYEKBKGBMAYPWM2DZQRCSR7CWC","short_pith_number":"pith:ZYEKBKGB","schema_version":"1.0","canonical_sha256":"ce08a0a8c16030fb3343cc222947e2b0a3b18b52751445fd9abe71973a6cef12","source":{"kind":"arxiv","id":"1908.08530","version":4},"attestation_state":"computed","paper":{"title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bin Li, Furu Wei, Jifeng Dai, Lewei Lu, Weijie Su, Xizhou Zhu, Yue Cao","submitted_at":"2019-08-22T17:59:30Z","abstract_excerpt":"We introduce a new pre-trainable generic representation for visual-linguistic tasks, called Visual-Linguistic BERT (VL-BERT for short). VL-BERT adopts the simple yet powerful Transformer model as the backbone, and extends it to take both visual and linguistic embedded features as input. In it, each element of the input is either of a word from the input sentence, or a region-of-interest (RoI) from the input image. It is designed to fit for most of the visual-linguistic downstream tasks. To better exploit the generic representation, we pre-train VL-BERT on the massive-scale Conceptual Captions "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.08530","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-08-22T17:59:30Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"0083e73e044c3563939efcfed574bcc67744ab46b33c853c4812c205fc340f62","abstract_canon_sha256":"88df91684b1c297540a32d6c18bce8fca550ab7f1b45bece0dd0172b959e1e79"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:41:12.834319Z","signature_b64":"SmUTKKmKSUR1zXOJo7YizeBfRyX/hV7u2jn3vk7Yssm2cMCQUS6ttwagHBSB2K2dpGoTfc0dldTzKSW3yVESDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce08a0a8c16030fb3343cc222947e2b0a3b18b52751445fd9abe71973a6cef12","last_reissued_at":"2026-07-05T00:41:12.833892Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:41:12.833892Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bin Li, Furu Wei, Jifeng Dai, Lewei Lu, Weijie Su, Xizhou Zhu, Yue Cao","submitted_at":"2019-08-22T17:59:30Z","abstract_excerpt":"We introduce a new pre-trainable generic representation for visual-linguistic tasks, called Visual-Linguistic BERT (VL-BERT for short). VL-BERT adopts the simple yet powerful Transformer model as the backbone, and extends it to take both visual and linguistic embedded features as input. In it, each element of the input is either of a word from the input sentence, or a region-of-interest (RoI) from the input image. It is designed to fit for most of the visual-linguistic downstream tasks. To better exploit the generic representation, we pre-train VL-BERT on the massive-scale Conceptual Captions "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.08530","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.08530/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.08530","created_at":"2026-07-05T00:41:12.833957+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.08530v4","created_at":"2026-07-05T00:41:12.833957+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.08530","created_at":"2026-07-05T00:41:12.833957+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYEKBKGBMAYP","created_at":"2026-07-05T00:41:12.833957+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYEKBKGBMAYPWM2D","created_at":"2026-07-05T00:41:12.833957+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYEKBKGB","created_at":"2026-07-05T00:41:12.833957+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06097","citing_title":"PVCap: Towards Accurate 3D Dense Captioning via PseudoCap and VoxelCapNet","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01743","citing_title":"InterCMDM: Block-Causal Diffusion for Autoregressive Human Interaction Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11853","citing_title":"Task-Aware Structured Memory for Dynamic Multi-modal In-Context Learning","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2406.09250","citing_title":"MirrorCheck: Efficient Adversarial Defense for Vision-Language Models","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21326","citing_title":"MiMIC: Mitigating Visual Modality Collapse in Universal Multimodal Retrieval While Avoiding Semantic Misalignment","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC","json":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC.json","graph_json":"https://pith.science/api/pith-number/ZYEKBKGBMAYPWM2DZQRCSR7CWC/graph.json","events_json":"https://pith.science/api/pith-number/ZYEKBKGBMAYPWM2DZQRCSR7CWC/events.json","paper":"https://pith.science/paper/ZYEKBKGB"},"agent_actions":{"view_html":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC","download_json":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC.json","view_paper":"https://pith.science/paper/ZYEKBKGB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.08530&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYEKBKGBMAYPWM2DZQRCSR7CWC/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYEKBKGBMAYPWM2DZQRCSR7CWC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC/action/storage_attestation","attest_author":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC/action/author_attestation","sign_citation":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC/action/citation_signature","submit_replication":"https://pith.science/pith/ZYEKBKGBMAYPWM2DZQRCSR7CWC/action/replication_record"}},"created_at":"2026-07-05T00:41:12.833957+00:00","updated_at":"2026-07-05T00:41:12.833957+00:00"}