{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HNKNNFZQMKGW746FR5S5KM23RY","short_pith_number":"pith:HNKNNFZQ","schema_version":"1.0","canonical_sha256":"3b54d69730628d6ff3c58f65d5335b8e04f815593154cdb09de2d0cfe76e1336","source":{"kind":"arxiv","id":"2303.09166","version":1},"attestation_state":"computed","paper":{"title":"Identifiability Results for Multimodal Contrastive Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Marx, Alice Bizeul, Emanuele Palumbo, Imant Daunhawer, Julia E. Vogt","submitted_at":"2023-03-16T09:14:26Z","abstract_excerpt":"Contrastive learning is a cornerstone underlying recent progress in multi-view and multimodal learning, e.g., in representation learning with image/caption pairs. While its effectiveness is not yet fully understood, a line of recent work reveals that contrastive learning can invert the data generating process and recover ground truth latent factors shared between views. In this work, we present new identifiability results for multimodal contrastive learning, showing that it is possible to recover shared factors in a more general setup than the multi-view setting studied previously. Specificall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.09166","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-16T09:14:26Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"f38193a689af0d240785f42ed1ff568a71af9ebb5b4b60d6b94c7f31f5487777","abstract_canon_sha256":"ee36d9d5110d3b0663741daa4f1d917f511477ae72759e79348fbde8f68572cf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:49.627152Z","signature_b64":"txZXXWtDoIoVqyk10eH6HJjQpsDwd1afGgx98iGmL6RkuXKLP2+JFrM5E34kGxaZzUa1EzMAYrpZWqXW0xpuDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b54d69730628d6ff3c58f65d5335b8e04f815593154cdb09de2d0cfe76e1336","last_reissued_at":"2026-07-05T05:51:49.626540Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:49.626540Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Identifiability Results for Multimodal Contrastive Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Marx, Alice Bizeul, Emanuele Palumbo, Imant Daunhawer, Julia E. Vogt","submitted_at":"2023-03-16T09:14:26Z","abstract_excerpt":"Contrastive learning is a cornerstone underlying recent progress in multi-view and multimodal learning, e.g., in representation learning with image/caption pairs. While its effectiveness is not yet fully understood, a line of recent work reveals that contrastive learning can invert the data generating process and recover ground truth latent factors shared between views. In this work, we present new identifiability results for multimodal contrastive learning, showing that it is possible to recover shared factors in a more general setup than the multi-view setting studied previously. Specificall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.09166","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.09166/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.09166","created_at":"2026-07-05T05:51:49.626614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.09166v1","created_at":"2026-07-05T05:51:49.626614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.09166","created_at":"2026-07-05T05:51:49.626614+00:00"},{"alias_kind":"pith_short_12","alias_value":"HNKNNFZQMKGW","created_at":"2026-07-05T05:51:49.626614+00:00"},{"alias_kind":"pith_short_16","alias_value":"HNKNNFZQMKGW746F","created_at":"2026-07-05T05:51:49.626614+00:00"},{"alias_kind":"pith_short_8","alias_value":"HNKNNFZQ","created_at":"2026-07-05T05:51:49.626614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00858","citing_title":"MoVA: Learning Asymmetric Dual Projections for Modular Long Video-Text Alignment","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00273","citing_title":"When Do Diffusion Models learn to Generate Multiple Objects?","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03517","citing_title":"Understanding Self-Supervised Learning via Latent Distribution Matching","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19594","citing_title":"Unsupervised Causal Abstractions Discovery","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03517","citing_title":"Understanding Self-Supervised Learning via Latent Distribution Matching","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22196","citing_title":"Mechanistic Independence: A Principle for Identifiable Disentangled Representations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03517","citing_title":"Understanding Self-Supervised Learning via Latent Distribution Matching","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY","json":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY.json","graph_json":"https://pith.science/api/pith-number/HNKNNFZQMKGW746FR5S5KM23RY/graph.json","events_json":"https://pith.science/api/pith-number/HNKNNFZQMKGW746FR5S5KM23RY/events.json","paper":"https://pith.science/paper/HNKNNFZQ"},"agent_actions":{"view_html":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY","download_json":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY.json","view_paper":"https://pith.science/paper/HNKNNFZQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.09166&json=true","fetch_graph":"https://pith.science/api/pith-number/HNKNNFZQMKGW746FR5S5KM23RY/graph.json","fetch_events":"https://pith.science/api/pith-number/HNKNNFZQMKGW746FR5S5KM23RY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY/action/storage_attestation","attest_author":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY/action/author_attestation","sign_citation":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY/action/citation_signature","submit_replication":"https://pith.science/pith/HNKNNFZQMKGW746FR5S5KM23RY/action/replication_record"}},"created_at":"2026-07-05T05:51:49.626614+00:00","updated_at":"2026-07-05T05:51:49.626614+00:00"}