{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LGKCLSOFCYHVVFAWLQOTYHFBLD","short_pith_number":"pith:LGKCLSOF","schema_version":"1.0","canonical_sha256":"599425c9c5160f5a94165c1d3c1ca158f2eb5bca31d56ea747bef251e298e970","source":{"kind":"arxiv","id":"2108.10904","version":3},"attestation_state":"computed","paper":{"title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adams Wei Yu, Jiahui Yu, Yuan Cao, Yulia Tsvetkov, Zihang Dai, Zirui Wang","submitted_at":"2021-08-24T18:14:00Z","abstract_excerpt":"With recent progress in joint modeling of visual and textual representations, Vision-Language Pretraining (VLP) has achieved impressive performance on many multimodal downstream tasks. However, the requirement for expensive annotations including clean image captions and regional labels limits the scalability of existing approaches, and complicates the pretraining procedure with the introduction of multiple dataset-specific objectives. In this work, we relax these constraints and present a minimalist pretraining framework, named Simple Visual Language Model (SimVLM). Unlike prior work, SimVLM r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.10904","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-08-24T18:14:00Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"14f7e191610cf6eb62e22d39a370a750b99d1e74c66b584738abb1b6c9208078","abstract_canon_sha256":"883079dbc073bc84bc29f4716951961068b7c8b9412e3b0cf91cc63bc537b589"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:23:21.390834Z","signature_b64":"hgLczYIzNtkyGamH1lDfpyM62ktvqSRj4GB0pWNr5pkLFeOMkjxB51MhpgeBXE0RQdZJJQLglSB/omvrFOPpDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"599425c9c5160f5a94165c1d3c1ca158f2eb5bca31d56ea747bef251e298e970","last_reissued_at":"2026-07-05T04:23:21.390422Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:23:21.390422Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adams Wei Yu, Jiahui Yu, Yuan Cao, Yulia Tsvetkov, Zihang Dai, Zirui Wang","submitted_at":"2021-08-24T18:14:00Z","abstract_excerpt":"With recent progress in joint modeling of visual and textual representations, Vision-Language Pretraining (VLP) has achieved impressive performance on many multimodal downstream tasks. However, the requirement for expensive annotations including clean image captions and regional labels limits the scalability of existing approaches, and complicates the pretraining procedure with the introduction of multiple dataset-specific objectives. In this work, we relax these constraints and present a minimalist pretraining framework, named Simple Visual Language Model (SimVLM). Unlike prior work, SimVLM r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.10904","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.10904/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.10904","created_at":"2026-07-05T04:23:21.390476+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.10904v3","created_at":"2026-07-05T04:23:21.390476+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.10904","created_at":"2026-07-05T04:23:21.390476+00:00"},{"alias_kind":"pith_short_12","alias_value":"LGKCLSOFCYHV","created_at":"2026-07-05T04:23:21.390476+00:00"},{"alias_kind":"pith_short_16","alias_value":"LGKCLSOFCYHVVFAW","created_at":"2026-07-05T04:23:21.390476+00:00"},{"alias_kind":"pith_short_8","alias_value":"LGKCLSOF","created_at":"2026-07-05T04:23:21.390476+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25298","citing_title":"KidRisk: Benchmark Dataset for Children Dangerous Action Recognition","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18147","citing_title":"WEQA: Wearable hEalth Question Answering with Query-Adaptive Agentic Reasoning","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12633","citing_title":"ECA: Efficient Continual Alignment for Open-Ended Image-to-Text Generation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29431","citing_title":"FADE: Mitigating Hallucinations by Reducing Language-Prior Dominance in Large Vision-Language Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00809","citing_title":"Let ViT Speak: Generative Language-Image Pre-training","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24020","citing_title":"Machine Intelligence that Understands Visual and Linguistic Information and Interacts with Humans and Environments","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29431","citing_title":"FADE: Mitigating Hallucinations by Reducing Language-Prior Dominance in Large Vision-Language Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2401.03568","citing_title":"Agent AI: Surveying the Horizons of Multimodal Interaction","ref_index":290,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02271","citing_title":"Medical Report Generation: A Hierarchical Task Structure-Based Cross-Modal Causal Intervention Framework","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2204.00598","citing_title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2111.11432","citing_title":"Florence: A New Foundation Model for Computer Vision","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2209.06794","citing_title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2205.01917","citing_title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2303.16199","citing_title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2205.06175","citing_title":"A Generalist Agent","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27559","citing_title":"RIHA: Report-Image Hierarchical Alignment for Radiology Report Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2206.10789","citing_title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05608","citing_title":"Inner Monologue: Embodied Reasoning through Planning with Language Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00809","citing_title":"Let ViT Speak: Generative Language-Image Pre-training","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07726","citing_title":"PaliGemma: A versatile 3B VLM for transfer","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05583","citing_title":"WRF4CIR: Weight-Regularized Fine-Tuning Network for Composed Image Retrieval","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13970","citing_title":"MApLe: Multi-instance Alignment of Diagnostic Reports and Large Medical Images","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD","json":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD.json","graph_json":"https://pith.science/api/pith-number/LGKCLSOFCYHVVFAWLQOTYHFBLD/graph.json","events_json":"https://pith.science/api/pith-number/LGKCLSOFCYHVVFAWLQOTYHFBLD/events.json","paper":"https://pith.science/paper/LGKCLSOF"},"agent_actions":{"view_html":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD","download_json":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD.json","view_paper":"https://pith.science/paper/LGKCLSOF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.10904&json=true","fetch_graph":"https://pith.science/api/pith-number/LGKCLSOFCYHVVFAWLQOTYHFBLD/graph.json","fetch_events":"https://pith.science/api/pith-number/LGKCLSOFCYHVVFAWLQOTYHFBLD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD/action/storage_attestation","attest_author":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD/action/author_attestation","sign_citation":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD/action/citation_signature","submit_replication":"https://pith.science/pith/LGKCLSOFCYHVVFAWLQOTYHFBLD/action/replication_record"}},"created_at":"2026-07-05T04:23:21.390476+00:00","updated_at":"2026-07-05T04:23:21.390476+00:00"}