{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:JUD7TLHYSPZ4S6STUD4LNYDVY4","short_pith_number":"pith:JUD7TLHY","schema_version":"1.0","canonical_sha256":"4d07f9acf893f3c97a53a0f8b6e075c711434958142f9fdf02f4709afedd36e3","source":{"kind":"arxiv","id":"2106.12570","version":3},"attestation_state":"computed","paper":{"title":"Learning Multimodal VAEs through Mutual Supervision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"N. Siddharth, Philip H.S. Torr, Sebastian M. Schmon, Tom Joy, Tom Rainforth, Yuge Shi","submitted_at":"2021-06-23T17:54:35Z","abstract_excerpt":"Multimodal VAEs seek to model the joint distribution over heterogeneous data (e.g.\\ vision, language), whilst also capturing a shared representation across such modalities. Prior work has typically combined information from the modalities by reconciling idiosyncratic representations directly in the recognition model through explicit products, mixtures, or other such factorisations. Here we introduce a novel alternative, the MEME, that avoids such explicit combinations by repurposing semi-supervised VAEs to combine information between modalities implicitly through mutual supervision. This formu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.12570","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-23T17:54:35Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"8b9cd447edeae279eebf69c4e5eddc4cee948540721070571f12557f833923d6","abstract_canon_sha256":"5719fd60727938d770281ed7993c8a8ea9ba7fbc5045611f165264e1159cfd8d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:25:46.885713Z","signature_b64":"RCfp748w++52XQCNJedXM+cJ0SxawxUzKt1m8HgrF8BBAnbEVuWFV8jyWQV7Hn7ZeCSF7QwaHNC+R0Wr58v5Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d07f9acf893f3c97a53a0f8b6e075c711434958142f9fdf02f4709afedd36e3","last_reissued_at":"2026-07-05T05:25:46.885239Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:25:46.885239Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Multimodal VAEs through Mutual Supervision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"N. Siddharth, Philip H.S. Torr, Sebastian M. Schmon, Tom Joy, Tom Rainforth, Yuge Shi","submitted_at":"2021-06-23T17:54:35Z","abstract_excerpt":"Multimodal VAEs seek to model the joint distribution over heterogeneous data (e.g.\\ vision, language), whilst also capturing a shared representation across such modalities. Prior work has typically combined information from the modalities by reconciling idiosyncratic representations directly in the recognition model through explicit products, mixtures, or other such factorisations. Here we introduce a novel alternative, the MEME, that avoids such explicit combinations by repurposing semi-supervised VAEs to combine information between modalities implicitly through mutual supervision. This formu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.12570","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.12570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.12570","created_at":"2026-07-05T05:25:46.885295+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.12570v3","created_at":"2026-07-05T05:25:46.885295+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.12570","created_at":"2026-07-05T05:25:46.885295+00:00"},{"alias_kind":"pith_short_12","alias_value":"JUD7TLHYSPZ4","created_at":"2026-07-05T05:25:46.885295+00:00"},{"alias_kind":"pith_short_16","alias_value":"JUD7TLHYSPZ4S6ST","created_at":"2026-07-05T05:25:46.885295+00:00"},{"alias_kind":"pith_short_8","alias_value":"JUD7TLHY","created_at":"2026-07-05T05:25:46.885295+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.21670","citing_title":"Diverse via bounded Agreement: Geometric Regularization for Multimodal Fusion","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4","json":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4.json","graph_json":"https://pith.science/api/pith-number/JUD7TLHYSPZ4S6STUD4LNYDVY4/graph.json","events_json":"https://pith.science/api/pith-number/JUD7TLHYSPZ4S6STUD4LNYDVY4/events.json","paper":"https://pith.science/paper/JUD7TLHY"},"agent_actions":{"view_html":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4","download_json":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4.json","view_paper":"https://pith.science/paper/JUD7TLHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.12570&json=true","fetch_graph":"https://pith.science/api/pith-number/JUD7TLHYSPZ4S6STUD4LNYDVY4/graph.json","fetch_events":"https://pith.science/api/pith-number/JUD7TLHYSPZ4S6STUD4LNYDVY4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4/action/storage_attestation","attest_author":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4/action/author_attestation","sign_citation":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4/action/citation_signature","submit_replication":"https://pith.science/pith/JUD7TLHYSPZ4S6STUD4LNYDVY4/action/replication_record"}},"created_at":"2026-07-05T05:25:46.885295+00:00","updated_at":"2026-07-05T05:25:46.885295+00:00"}