{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7PDN4I4XHMWJWYILGGFFSZLKFJ","short_pith_number":"pith:7PDN4I4X","schema_version":"1.0","canonical_sha256":"fbc6de23973b2c9b610b318a59656a2a7be7e3086bb4bb1c5d1573a20682cb4a","source":{"kind":"arxiv","id":"2311.18021","version":2},"attestation_state":"computed","paper":{"title":"Can Multimodal Large Language Models Truly Perform Multimodal In-Context Learning?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bailan He, Jianzhe Liu, Jindong Gu, Mark Buckley, Philip Torr, Shuo Chen, Volker Tresp, Yao Qin, Zhen Han","submitted_at":"2023-11-29T19:08:11Z","abstract_excerpt":"Large Language Models (LLMs) with in-context learning (ICL) ability can quickly adapt to a specific context given a few demonstrations (demos). Recently, Multimodal Large Language Models (MLLMs) built upon LLMs have also shown multimodal ICL ability, i.e., responding to queries given a few multimodal demos, including images, queries, and answers. While ICL has been extensively studied on LLMs, its research on MLLMs remains limited. One essential question is whether these MLLMs can truly conduct multimodal ICL, or if only the textual modality is necessary. We investigate this question by examin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.18021","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-29T19:08:11Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"12be4391773e896bf8d3bee6d2ce7c253ba406b496bd892e80889af1b4789e0b","abstract_canon_sha256":"b9373e6825bcec4dac36c13eacb9d40ffe5e6eb7fe90d784e6c4ffc13595010b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:42.255061Z","signature_b64":"FrTSdLWxw/hMqmgo/DeFJAaoP7BUZat0x+UB9rfmxxuOVczEvt026Ih/LXWuUKm9/0gHKnWkdSNp5qcUy1lTAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fbc6de23973b2c9b610b318a59656a2a7be7e3086bb4bb1c5d1573a20682cb4a","last_reissued_at":"2026-07-05T09:45:42.254591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:42.254591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Multimodal Large Language Models Truly Perform Multimodal In-Context Learning?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bailan He, Jianzhe Liu, Jindong Gu, Mark Buckley, Philip Torr, Shuo Chen, Volker Tresp, Yao Qin, Zhen Han","submitted_at":"2023-11-29T19:08:11Z","abstract_excerpt":"Large Language Models (LLMs) with in-context learning (ICL) ability can quickly adapt to a specific context given a few demonstrations (demos). Recently, Multimodal Large Language Models (MLLMs) built upon LLMs have also shown multimodal ICL ability, i.e., responding to queries given a few multimodal demos, including images, queries, and answers. While ICL has been extensively studied on LLMs, its research on MLLMs remains limited. One essential question is whether these MLLMs can truly conduct multimodal ICL, or if only the textual modality is necessary. We investigate this question by examin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.18021","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.18021/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.18021","created_at":"2026-07-05T09:45:42.254647+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.18021v2","created_at":"2026-07-05T09:45:42.254647+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.18021","created_at":"2026-07-05T09:45:42.254647+00:00"},{"alias_kind":"pith_short_12","alias_value":"7PDN4I4XHMWJ","created_at":"2026-07-05T09:45:42.254647+00:00"},{"alias_kind":"pith_short_16","alias_value":"7PDN4I4XHMWJWYIL","created_at":"2026-07-05T09:45:42.254647+00:00"},{"alias_kind":"pith_short_8","alias_value":"7PDN4I4X","created_at":"2026-07-05T09:45:42.254647+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.18117","citing_title":"Online In-Context Distillation for Low-Resource Vision Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13403","citing_title":"Why Multimodal In-Context Learning Lags Behind? Unveiling the Inner Mechanisms and Bottlenecks","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ","json":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ.json","graph_json":"https://pith.science/api/pith-number/7PDN4I4XHMWJWYILGGFFSZLKFJ/graph.json","events_json":"https://pith.science/api/pith-number/7PDN4I4XHMWJWYILGGFFSZLKFJ/events.json","paper":"https://pith.science/paper/7PDN4I4X"},"agent_actions":{"view_html":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ","download_json":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ.json","view_paper":"https://pith.science/paper/7PDN4I4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.18021&json=true","fetch_graph":"https://pith.science/api/pith-number/7PDN4I4XHMWJWYILGGFFSZLKFJ/graph.json","fetch_events":"https://pith.science/api/pith-number/7PDN4I4XHMWJWYILGGFFSZLKFJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ/action/storage_attestation","attest_author":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ/action/author_attestation","sign_citation":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ/action/citation_signature","submit_replication":"https://pith.science/pith/7PDN4I4XHMWJWYILGGFFSZLKFJ/action/replication_record"}},"created_at":"2026-07-05T09:45:42.254647+00:00","updated_at":"2026-07-05T09:45:42.254647+00:00"}