{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MPMKZ4YD4LN656YYVWJV7LTIFJ","short_pith_number":"pith:MPMKZ4YD","schema_version":"1.0","canonical_sha256":"63d8acf303e2dbeefb18ad935fae682a5a283a913bc47e1b19bc7070929e355e","source":{"kind":"arxiv","id":"2407.00118","version":1},"attestation_state":"computed","paper":{"title":"From Efficient Multimodal Models to World Models: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Haoran Wang, Junxiong Lin, Wenqiang Zhang, Xinji Mai, Yang Chang, Yanlan Kang, Yan Wang, Zeng Tao","submitted_at":"2024-06-27T15:36:43Z","abstract_excerpt":"Multimodal Large Models (MLMs) are becoming a significant research focus, combining powerful large language models with multimodal learning to perform complex tasks across different data modalities. This review explores the latest developments and challenges in MLMs, emphasizing their potential in achieving artificial general intelligence and as a pathway to world models. We provide an overview of key techniques such as Multimodal Chain of Thought (M-COT), Multimodal Instruction Tuning (M-IT), and Multimodal In-Context Learning (M-ICL). Additionally, we discuss both the fundamental and specifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00118","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-27T15:36:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b76aed7b9ea3b1d9c7c945909d45deda440f72933c905b9ddae6fdec119e48c0","abstract_canon_sha256":"cc393873a7f8a98ac96402fc02cb5ddb7c614cdba1f334223c89ba408cc59e99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:14.686452Z","signature_b64":"IhY0bNvBZKnCWN5W9Djt/D7oNg1tkEnK4SceHpQ8rP9xNz5vs+BVBVdrXQq/o/1lDIEY7aejr+j/23YMLxbwCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"63d8acf303e2dbeefb18ad935fae682a5a283a913bc47e1b19bc7070929e355e","last_reissued_at":"2026-07-05T08:38:14.686009Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:14.686009Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Efficient Multimodal Models to World Models: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Haoran Wang, Junxiong Lin, Wenqiang Zhang, Xinji Mai, Yang Chang, Yanlan Kang, Yan Wang, Zeng Tao","submitted_at":"2024-06-27T15:36:43Z","abstract_excerpt":"Multimodal Large Models (MLMs) are becoming a significant research focus, combining powerful large language models with multimodal learning to perform complex tasks across different data modalities. This review explores the latest developments and challenges in MLMs, emphasizing their potential in achieving artificial general intelligence and as a pathway to world models. We provide an overview of key techniques such as Multimodal Chain of Thought (M-COT), Multimodal Instruction Tuning (M-IT), and Multimodal In-Context Learning (M-ICL). Additionally, we discuss both the fundamental and specifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00118","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00118/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00118","created_at":"2026-07-05T08:38:14.686070+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00118v1","created_at":"2026-07-05T08:38:14.686070+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00118","created_at":"2026-07-05T08:38:14.686070+00:00"},{"alias_kind":"pith_short_12","alias_value":"MPMKZ4YD4LN6","created_at":"2026-07-05T08:38:14.686070+00:00"},{"alias_kind":"pith_short_16","alias_value":"MPMKZ4YD4LN656YY","created_at":"2026-07-05T08:38:14.686070+00:00"},{"alias_kind":"pith_short_8","alias_value":"MPMKZ4YD","created_at":"2026-07-05T08:38:14.686070+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27431","citing_title":"Tackling Multimodal Learning Challenges with Mixture-of-Expert: A Survey","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00133","citing_title":"World Models: A Comprehensive Survey of Architectures, Methodologies, Reasoning Paradigms, and Applications","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04978","citing_title":"Aligning Perception, Reasoning, Modeling and Interaction: A Survey on Physical AI","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07199","citing_title":"Three-in-One World Model: Energy-Based Consistency, Prediction, and Counterfactual Inference for Marketing Intervention","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ","json":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ.json","graph_json":"https://pith.science/api/pith-number/MPMKZ4YD4LN656YYVWJV7LTIFJ/graph.json","events_json":"https://pith.science/api/pith-number/MPMKZ4YD4LN656YYVWJV7LTIFJ/events.json","paper":"https://pith.science/paper/MPMKZ4YD"},"agent_actions":{"view_html":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ","download_json":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ.json","view_paper":"https://pith.science/paper/MPMKZ4YD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00118&json=true","fetch_graph":"https://pith.science/api/pith-number/MPMKZ4YD4LN656YYVWJV7LTIFJ/graph.json","fetch_events":"https://pith.science/api/pith-number/MPMKZ4YD4LN656YYVWJV7LTIFJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ/action/storage_attestation","attest_author":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ/action/author_attestation","sign_citation":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ/action/citation_signature","submit_replication":"https://pith.science/pith/MPMKZ4YD4LN656YYVWJV7LTIFJ/action/replication_record"}},"created_at":"2026-07-05T08:38:14.686070+00:00","updated_at":"2026-07-05T08:38:14.686070+00:00"}