{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:B35VIPLI2BNQF2TH4A26MSV3ZM","short_pith_number":"pith:B35VIPLI","schema_version":"1.0","canonical_sha256":"0efb543d68d05b02ea67e035e64abbcb17c294901fc7f398f2ed79b8115d73e9","source":{"kind":"arxiv","id":"2312.00935","version":2},"attestation_state":"computed","paper":{"title":"Understanding Unimodal Bias in Multimodal Deep Linear Networks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Saxe, Peter E. Latham, Yedi Zhang","submitted_at":"2023-12-01T21:29:54Z","abstract_excerpt":"Using multiple input streams simultaneously to train multimodal neural networks is intuitively advantageous but practically challenging. A key challenge is unimodal bias, where a network overly relies on one modality and ignores others during joint training. We develop a theory of unimodal bias with multimodal deep linear networks to understand how architecture and data statistics influence this bias. This is the first work to calculate the duration of the unimodal phase in learning as a function of the depth at which modalities are fused within the network, dataset statistics, and initializat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.00935","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-01T21:29:54Z","cross_cats_sorted":[],"title_canon_sha256":"1564d4ac1b11c74af2f675d07081393ed58da83c9e4256c1b8c68d0dc35ee54b","abstract_canon_sha256":"343d2da7d01f52936281a28763218577985a1ad1d77214e6bc12bd0b063d7921"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:27.119235Z","signature_b64":"UGgd0MrgTXXFoxSa45wRNvTmWhZM9gRwi2Dv+5Ez4OYqW1PASPAMZdNpArR51grE/sHecMcH7rQya+L7g9XzDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0efb543d68d05b02ea67e035e64abbcb17c294901fc7f398f2ed79b8115d73e9","last_reissued_at":"2026-07-05T08:49:27.118681Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:27.118681Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Unimodal Bias in Multimodal Deep Linear Networks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Saxe, Peter E. Latham, Yedi Zhang","submitted_at":"2023-12-01T21:29:54Z","abstract_excerpt":"Using multiple input streams simultaneously to train multimodal neural networks is intuitively advantageous but practically challenging. A key challenge is unimodal bias, where a network overly relies on one modality and ignores others during joint training. We develop a theory of unimodal bias with multimodal deep linear networks to understand how architecture and data statistics influence this bias. This is the first work to calculate the duration of the unimodal phase in learning as a function of the depth at which modalities are fused within the network, dataset statistics, and initializat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.00935","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.00935/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.00935","created_at":"2026-07-05T08:49:27.118741+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.00935v2","created_at":"2026-07-05T08:49:27.118741+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.00935","created_at":"2026-07-05T08:49:27.118741+00:00"},{"alias_kind":"pith_short_12","alias_value":"B35VIPLI2BNQ","created_at":"2026-07-05T08:49:27.118741+00:00"},{"alias_kind":"pith_short_16","alias_value":"B35VIPLI2BNQF2TH","created_at":"2026-07-05T08:49:27.118741+00:00"},{"alias_kind":"pith_short_8","alias_value":"B35VIPLI","created_at":"2026-07-05T08:49:27.118741+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28869","citing_title":"Balancing Multimodal Learning through Label Space Reshaping","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM","json":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM.json","graph_json":"https://pith.science/api/pith-number/B35VIPLI2BNQF2TH4A26MSV3ZM/graph.json","events_json":"https://pith.science/api/pith-number/B35VIPLI2BNQF2TH4A26MSV3ZM/events.json","paper":"https://pith.science/paper/B35VIPLI"},"agent_actions":{"view_html":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM","download_json":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM.json","view_paper":"https://pith.science/paper/B35VIPLI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.00935&json=true","fetch_graph":"https://pith.science/api/pith-number/B35VIPLI2BNQF2TH4A26MSV3ZM/graph.json","fetch_events":"https://pith.science/api/pith-number/B35VIPLI2BNQF2TH4A26MSV3ZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM/action/storage_attestation","attest_author":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM/action/author_attestation","sign_citation":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM/action/citation_signature","submit_replication":"https://pith.science/pith/B35VIPLI2BNQF2TH4A26MSV3ZM/action/replication_record"}},"created_at":"2026-07-05T08:49:27.118741+00:00","updated_at":"2026-07-05T08:49:27.118741+00:00"}