{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YQJRKA5LWCDPLRQ5HGQS35QZZL","short_pith_number":"pith:YQJRKA5L","schema_version":"1.0","canonical_sha256":"c4131503abb086f5c61d39a12df619cac178ad74539a9c109502725a2deb1080","source":{"kind":"arxiv","id":"2405.17927","version":1},"attestation_state":"computed","paper":{"title":"The Evolution of Multimodal Model Architectures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG","eess.AS"],"primary_cat":"cs.AI","authors_text":"Abhishek Chaurasia, Aman Chadha, Eugenio Culurciello, Shakti N. Wadekar","submitted_at":"2024-05-28T07:48:15Z","abstract_excerpt":"This work uniquely identifies and characterizes four prevalent multimodal model architectural patterns in the contemporary multimodal landscape. Systematically categorizing models by architecture type facilitates monitoring of developments in the multimodal domain. Distinct from recent survey papers that present general information on multimodal architectures, this research conducts a comprehensive exploration of architectural details and identifies four specific architectural types. The types are distinguished by their respective methodologies for integrating multimodal inputs into the deep n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17927","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-05-28T07:48:15Z","cross_cats_sorted":["cs.CL","cs.CV","cs.LG","eess.AS"],"title_canon_sha256":"9e261ed82bd1b0f68e39f3ddcadfef7448a7caea645da85cc8cec48307373fa3","abstract_canon_sha256":"b0f3838d24d107957e6e08547d3275b26d16ac8035b91716761bf5636528de84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:13.241464Z","signature_b64":"KWPF+AorHpBF8SaL2IdyrEbZ69myLTo3UPc3IKbrXGe+VtcoGAM4VzG/3nPli4c6gc/jrIcR3uY3JgO48BAUCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4131503abb086f5c61d39a12df619cac178ad74539a9c109502725a2deb1080","last_reissued_at":"2026-07-05T08:24:13.240978Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:13.240978Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Evolution of Multimodal Model Architectures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.LG","eess.AS"],"primary_cat":"cs.AI","authors_text":"Abhishek Chaurasia, Aman Chadha, Eugenio Culurciello, Shakti N. Wadekar","submitted_at":"2024-05-28T07:48:15Z","abstract_excerpt":"This work uniquely identifies and characterizes four prevalent multimodal model architectural patterns in the contemporary multimodal landscape. Systematically categorizing models by architecture type facilitates monitoring of developments in the multimodal domain. Distinct from recent survey papers that present general information on multimodal architectures, this research conducts a comprehensive exploration of architectural details and identifies four specific architectural types. The types are distinguished by their respective methodologies for integrating multimodal inputs into the deep n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17927","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17927/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17927","created_at":"2026-07-05T08:24:13.241034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17927v1","created_at":"2026-07-05T08:24:13.241034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17927","created_at":"2026-07-05T08:24:13.241034+00:00"},{"alias_kind":"pith_short_12","alias_value":"YQJRKA5LWCDP","created_at":"2026-07-05T08:24:13.241034+00:00"},{"alias_kind":"pith_short_16","alias_value":"YQJRKA5LWCDPLRQ5","created_at":"2026-07-05T08:24:13.241034+00:00"},{"alias_kind":"pith_short_8","alias_value":"YQJRKA5L","created_at":"2026-07-05T08:24:13.241034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09985","citing_title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL","json":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL.json","graph_json":"https://pith.science/api/pith-number/YQJRKA5LWCDPLRQ5HGQS35QZZL/graph.json","events_json":"https://pith.science/api/pith-number/YQJRKA5LWCDPLRQ5HGQS35QZZL/events.json","paper":"https://pith.science/paper/YQJRKA5L"},"agent_actions":{"view_html":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL","download_json":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL.json","view_paper":"https://pith.science/paper/YQJRKA5L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17927&json=true","fetch_graph":"https://pith.science/api/pith-number/YQJRKA5LWCDPLRQ5HGQS35QZZL/graph.json","fetch_events":"https://pith.science/api/pith-number/YQJRKA5LWCDPLRQ5HGQS35QZZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL/action/storage_attestation","attest_author":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL/action/author_attestation","sign_citation":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL/action/citation_signature","submit_replication":"https://pith.science/pith/YQJRKA5LWCDPLRQ5HGQS35QZZL/action/replication_record"}},"created_at":"2026-07-05T08:24:13.241034+00:00","updated_at":"2026-07-05T08:24:13.241034+00:00"}