{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:GDYZIHN5UDNT6E4QUUGRA3PSFV","short_pith_number":"pith:GDYZIHN5","schema_version":"1.0","canonical_sha256":"30f1941dbda0db3f1390a50d106df22d45d0813ceab61b74db94fc62e6143285","source":{"kind":"arxiv","id":"2208.02131","version":2},"attestation_state":"computed","paper":{"title":"Masked Vision and Language Modeling for Multi-modal Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Avinash Ravichandran, Erhan Bas, Gukyeong Kwon, Rahul Bhotika, Stefano Soatto, Zhaowei Cai","submitted_at":"2022-08-03T15:11:01Z","abstract_excerpt":"In this paper, we study how to use masked signal modeling in vision and language (V+L) representation learning. Instead of developing masked language modeling (MLM) and masked image modeling (MIM) independently, we propose to build joint masked vision and language modeling, where the masked signal of one modality is reconstructed with the help from another modality. This is motivated by the nature of image-text paired data that both of the image and the text convey almost the same information but in different formats. The masked signal reconstruction of one modality conditioned on another moda"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.02131","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-08-03T15:11:01Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"f1679b132529d22928c4dcae8925f164e33d0c474ab909c309b44e50c7ada098","abstract_canon_sha256":"6206a5e05bdeff7d5cc0b57528bf7fb203a4328572965757c7eb6f952fa4c3a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:26.363192Z","signature_b64":"J7sui3jakAaQcSJoDm5Fqr8CmBGo5LWbfAbLKhqJoSTSF8rzmumQZFmYKjGLo8Qf323TtxrhvW9ejR9jThLrBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30f1941dbda0db3f1390a50d106df22d45d0813ceab61b74db94fc62e6143285","last_reissued_at":"2026-07-05T05:51:26.362814Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:26.362814Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Masked Vision and Language Modeling for Multi-modal Representation Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Avinash Ravichandran, Erhan Bas, Gukyeong Kwon, Rahul Bhotika, Stefano Soatto, Zhaowei Cai","submitted_at":"2022-08-03T15:11:01Z","abstract_excerpt":"In this paper, we study how to use masked signal modeling in vision and language (V+L) representation learning. Instead of developing masked language modeling (MLM) and masked image modeling (MIM) independently, we propose to build joint masked vision and language modeling, where the masked signal of one modality is reconstructed with the help from another modality. This is motivated by the nature of image-text paired data that both of the image and the text convey almost the same information but in different formats. The masked signal reconstruction of one modality conditioned on another moda"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.02131","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.02131/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.02131","created_at":"2026-07-05T05:51:26.362880+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.02131v2","created_at":"2026-07-05T05:51:26.362880+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.02131","created_at":"2026-07-05T05:51:26.362880+00:00"},{"alias_kind":"pith_short_12","alias_value":"GDYZIHN5UDNT","created_at":"2026-07-05T05:51:26.362880+00:00"},{"alias_kind":"pith_short_16","alias_value":"GDYZIHN5UDNT6E4Q","created_at":"2026-07-05T05:51:26.362880+00:00"},{"alias_kind":"pith_short_8","alias_value":"GDYZIHN5","created_at":"2026-07-05T05:51:26.362880+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13288","citing_title":"Cross-Modal Masked Compositional Concept Modeling for Enhancing Visio-Linguistic Compositionality","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2408.04840","citing_title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2311.04257","citing_title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03893","citing_title":"FeynmanBench: Benchmarking Multimodal LLMs on Diagrammatic Physics Reasoning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25255","citing_title":"Personalized Cross-Modal Emotional Correlation Learning for Speech-Preserving Facial Expression Manipulation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02283","citing_title":"Rethinking Electro-Optical Vision Foundation Models for Remote Sensing Retrieval: A Controlled Comparison with Generalist VFM","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV","json":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV.json","graph_json":"https://pith.science/api/pith-number/GDYZIHN5UDNT6E4QUUGRA3PSFV/graph.json","events_json":"https://pith.science/api/pith-number/GDYZIHN5UDNT6E4QUUGRA3PSFV/events.json","paper":"https://pith.science/paper/GDYZIHN5"},"agent_actions":{"view_html":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV","download_json":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV.json","view_paper":"https://pith.science/paper/GDYZIHN5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.02131&json=true","fetch_graph":"https://pith.science/api/pith-number/GDYZIHN5UDNT6E4QUUGRA3PSFV/graph.json","fetch_events":"https://pith.science/api/pith-number/GDYZIHN5UDNT6E4QUUGRA3PSFV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV/action/storage_attestation","attest_author":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV/action/author_attestation","sign_citation":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV/action/citation_signature","submit_replication":"https://pith.science/pith/GDYZIHN5UDNT6E4QUUGRA3PSFV/action/replication_record"}},"created_at":"2026-07-05T05:51:26.362880+00:00","updated_at":"2026-07-05T05:51:26.362880+00:00"}