{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:F3VXXHZHBKMHPVVQKP37BMMELT","short_pith_number":"pith:F3VXXHZH","schema_version":"1.0","canonical_sha256":"2eeb7b9f270a9877d6b053f7f0b1845cf4f95534d194884ba3c7c937a1874958","source":{"kind":"arxiv","id":"2011.07191","version":1},"attestation_state":"computed","paper":{"title":"On the Benefits of Early Fusion in Multimodal Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"George Barnum, Sabera Talukder, Yisong Yue","submitted_at":"2020-11-14T01:58:41Z","abstract_excerpt":"Intelligently reasoning about the world often requires integrating data from multiple modalities, as any individual modality may contain unreliable or incomplete information. Prior work in multimodal learning fuses input modalities only after significant independent processing. On the other hand, the brain performs multimodal processing almost immediately. This divide between conventional multimodal learning and neuroscience suggests that a detailed study of early multimodal fusion could improve artificial multimodal representations. To facilitate the study of early multimodal fusion, we creat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.07191","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2020-11-14T01:58:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4844d8086b1e4020047634ebb78987601e28442156bcb6e891c7df640107c920","abstract_canon_sha256":"78365a5792e15dc7becaabc2cc2146099b7be4a8df668e19032e23b7998a7af2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:51:29.912816Z","signature_b64":"h1Qs8cPRjwt1wC3mCdgEQ8zBfykIBaRfuh0PMsI/NNXGVPb2EnurPUs9r5We0xR0aBL+3vydrgBWVMMh9gNZCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2eeb7b9f270a9877d6b053f7f0b1845cf4f95534d194884ba3c7c937a1874958","last_reissued_at":"2026-07-05T01:51:29.912472Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:51:29.912472Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Benefits of Early Fusion in Multimodal Representation Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"George Barnum, Sabera Talukder, Yisong Yue","submitted_at":"2020-11-14T01:58:41Z","abstract_excerpt":"Intelligently reasoning about the world often requires integrating data from multiple modalities, as any individual modality may contain unreliable or incomplete information. Prior work in multimodal learning fuses input modalities only after significant independent processing. On the other hand, the brain performs multimodal processing almost immediately. This divide between conventional multimodal learning and neuroscience suggests that a detailed study of early multimodal fusion could improve artificial multimodal representations. To facilitate the study of early multimodal fusion, we creat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.07191","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.07191/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.07191","created_at":"2026-07-05T01:51:29.912528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.07191v1","created_at":"2026-07-05T01:51:29.912528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.07191","created_at":"2026-07-05T01:51:29.912528+00:00"},{"alias_kind":"pith_short_12","alias_value":"F3VXXHZHBKMH","created_at":"2026-07-05T01:51:29.912528+00:00"},{"alias_kind":"pith_short_16","alias_value":"F3VXXHZHBKMHPVVQ","created_at":"2026-07-05T01:51:29.912528+00:00"},{"alias_kind":"pith_short_8","alias_value":"F3VXXHZH","created_at":"2026-07-05T01:51:29.912528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06867","citing_title":"Multi-FRuGaL: Multimodal Flexible Redundancy-aware Decomposed Gated Learning for Cancer Diagnosis and Prognosis","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT","json":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT.json","graph_json":"https://pith.science/api/pith-number/F3VXXHZHBKMHPVVQKP37BMMELT/graph.json","events_json":"https://pith.science/api/pith-number/F3VXXHZHBKMHPVVQKP37BMMELT/events.json","paper":"https://pith.science/paper/F3VXXHZH"},"agent_actions":{"view_html":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT","download_json":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT.json","view_paper":"https://pith.science/paper/F3VXXHZH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.07191&json=true","fetch_graph":"https://pith.science/api/pith-number/F3VXXHZHBKMHPVVQKP37BMMELT/graph.json","fetch_events":"https://pith.science/api/pith-number/F3VXXHZHBKMHPVVQKP37BMMELT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT/action/storage_attestation","attest_author":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT/action/author_attestation","sign_citation":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT/action/citation_signature","submit_replication":"https://pith.science/pith/F3VXXHZHBKMHPVVQKP37BMMELT/action/replication_record"}},"created_at":"2026-07-05T01:51:29.912528+00:00","updated_at":"2026-07-05T01:51:29.912528+00:00"}