{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OCMIMK4N7JFGPZXJLKW3ALVWPW","short_pith_number":"pith:OCMIMK4N","schema_version":"1.0","canonical_sha256":"7098862b8dfa4a67e6e95aadb02eb67da2c348b9a47b2a263cabb4365dcaffe4","source":{"kind":"arxiv","id":"2405.09798","version":2},"attestation_state":"computed","paper":{"title":"Many-Shot In-Context Learning in Multimodal Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Andrew Y. Ng, Jeremy Irvin, Ji Hun Wang, Jonathan H. Chen, Muhammad Ahmed Chaudhry, Yixing Jiang","submitted_at":"2024-05-16T04:02:43Z","abstract_excerpt":"Large language models are effective at few-shot in-context learning (ICL). Recent advancements in multimodal foundation models have enabled unprecedentedly long context windows, presenting an opportunity to explore their capability to perform ICL with many more demonstrating examples. In this work, we evaluate the performance of multimodal foundation models scaling from few-shot to many-shot ICL. We benchmark GPT-4o and Gemini 1.5 Pro across 14 datasets spanning multiple domains (natural imagery, medical imagery, remote sensing, and molecular imagery) and tasks (image classification, visual QA"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.09798","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-16T04:02:43Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"f020cbdf4b1ef9391883ce00ea0a39d430ac5640863657f007e73e49507a7096","abstract_canon_sha256":"30d2778c54b4e6c64947ff93d51314475c5fb63b9ffa0dde4145f82405875885"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:15.143134Z","signature_b64":"TDizdriMKJXKlZr50LXf+0Dq3bVCDcNt7C5hXQkx+UvCGyvIQTVsZFOKj9+FP+oO7RdnTOGRcESZDn697rgMCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7098862b8dfa4a67e6e95aadb02eb67da2c348b9a47b2a263cabb4365dcaffe4","last_reissued_at":"2026-07-05T09:16:15.142685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:15.142685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Many-Shot In-Context Learning in Multimodal Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Andrew Y. Ng, Jeremy Irvin, Ji Hun Wang, Jonathan H. Chen, Muhammad Ahmed Chaudhry, Yixing Jiang","submitted_at":"2024-05-16T04:02:43Z","abstract_excerpt":"Large language models are effective at few-shot in-context learning (ICL). Recent advancements in multimodal foundation models have enabled unprecedentedly long context windows, presenting an opportunity to explore their capability to perform ICL with many more demonstrating examples. In this work, we evaluate the performance of multimodal foundation models scaling from few-shot to many-shot ICL. We benchmark GPT-4o and Gemini 1.5 Pro across 14 datasets spanning multiple domains (natural imagery, medical imagery, remote sensing, and molecular imagery) and tasks (image classification, visual QA"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.09798","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.09798/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.09798","created_at":"2026-07-05T09:16:15.142735+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.09798v2","created_at":"2026-07-05T09:16:15.142735+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.09798","created_at":"2026-07-05T09:16:15.142735+00:00"},{"alias_kind":"pith_short_12","alias_value":"OCMIMK4N7JFG","created_at":"2026-07-05T09:16:15.142735+00:00"},{"alias_kind":"pith_short_16","alias_value":"OCMIMK4N7JFGPZXJ","created_at":"2026-07-05T09:16:15.142735+00:00"},{"alias_kind":"pith_short_8","alias_value":"OCMIMK4N","created_at":"2026-07-05T09:16:15.142735+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23598","citing_title":"When Youth Enter the Algorithmic Wild: Discovering and Understanding Potentially Harmful Teen Videos on Douyin and Kwai","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01955","citing_title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2406.09411","citing_title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10936","citing_title":"Personal Visual Context Learning in Large Multimodal Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07895","citing_title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW","json":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW.json","graph_json":"https://pith.science/api/pith-number/OCMIMK4N7JFGPZXJLKW3ALVWPW/graph.json","events_json":"https://pith.science/api/pith-number/OCMIMK4N7JFGPZXJLKW3ALVWPW/events.json","paper":"https://pith.science/paper/OCMIMK4N"},"agent_actions":{"view_html":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW","download_json":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW.json","view_paper":"https://pith.science/paper/OCMIMK4N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.09798&json=true","fetch_graph":"https://pith.science/api/pith-number/OCMIMK4N7JFGPZXJLKW3ALVWPW/graph.json","fetch_events":"https://pith.science/api/pith-number/OCMIMK4N7JFGPZXJLKW3ALVWPW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW/action/storage_attestation","attest_author":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW/action/author_attestation","sign_citation":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW/action/citation_signature","submit_replication":"https://pith.science/pith/OCMIMK4N7JFGPZXJLKW3ALVWPW/action/replication_record"}},"created_at":"2026-07-05T09:16:15.142735+00:00","updated_at":"2026-07-05T09:16:15.142735+00:00"}