{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LSR2E3ISP5LAY2TT3QBKSLWU6B","short_pith_number":"pith:LSR2E3IS","schema_version":"1.0","canonical_sha256":"5ca3a26d127f560c6a73dc02a92ed4f07a603ad38e6558bebeca478b74da4e22","source":{"kind":"arxiv","id":"2310.00647","version":2},"attestation_state":"computed","paper":{"title":"Beyond Task Performance: Evaluating and Reducing the Flaws of Large Multimodal Models with In-Context Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Alexandre Rame, Corentin Dancette, Matthieu Cord, Mustafa Shukor","submitted_at":"2023-10-01T12:02:59Z","abstract_excerpt":"Following the success of Large Language Models (LLMs), Large Multimodal Models (LMMs), such as the Flamingo model and its subsequent competitors, have started to emerge as natural steps towards generalist agents. However, interacting with recent LMMs reveals major limitations that are hardly captured by the current evaluation benchmarks. Indeed, task performances (e.g., VQA accuracy) alone do not provide enough clues to understand their real capabilities, limitations, and to which extent such models are aligned to human expectations. To refine our understanding of those flaws, we deviate from "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00647","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-01T12:02:59Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"48b2a39392920bbc837720e26a003ed2b3f3fe2af5f2ec94d191ee3530644d98","abstract_canon_sha256":"cad2663cbcde5d3e26a840c0dcf54923f102fe07d54a0b8d917f7f801680ac61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:36:03.391476Z","signature_b64":"TURCEIThVu8ZjHPnmNl24bbGRfuDRuynOPdHBTG6+Z5+b1Fj8mUASA6g/I3H4Ap75FKHbmoUXH95d64d6qi+Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ca3a26d127f560c6a73dc02a92ed4f07a603ad38e6558bebeca478b74da4e22","last_reissued_at":"2026-07-05T07:36:03.390974Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:36:03.390974Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Task Performance: Evaluating and Reducing the Flaws of Large Multimodal Models with In-Context Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Alexandre Rame, Corentin Dancette, Matthieu Cord, Mustafa Shukor","submitted_at":"2023-10-01T12:02:59Z","abstract_excerpt":"Following the success of Large Language Models (LLMs), Large Multimodal Models (LMMs), such as the Flamingo model and its subsequent competitors, have started to emerge as natural steps towards generalist agents. However, interacting with recent LMMs reveals major limitations that are hardly captured by the current evaluation benchmarks. Indeed, task performances (e.g., VQA accuracy) alone do not provide enough clues to understand their real capabilities, limitations, and to which extent such models are aligned to human expectations. To refine our understanding of those flaws, we deviate from "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00647","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00647","created_at":"2026-07-05T07:36:03.391036+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00647v2","created_at":"2026-07-05T07:36:03.391036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00647","created_at":"2026-07-05T07:36:03.391036+00:00"},{"alias_kind":"pith_short_12","alias_value":"LSR2E3ISP5LA","created_at":"2026-07-05T07:36:03.391036+00:00"},{"alias_kind":"pith_short_16","alias_value":"LSR2E3ISP5LAY2TT","created_at":"2026-07-05T07:36:03.391036+00:00"},{"alias_kind":"pith_short_8","alias_value":"LSR2E3IS","created_at":"2026-07-05T07:36:03.391036+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2407.07895","citing_title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B","json":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B.json","graph_json":"https://pith.science/api/pith-number/LSR2E3ISP5LAY2TT3QBKSLWU6B/graph.json","events_json":"https://pith.science/api/pith-number/LSR2E3ISP5LAY2TT3QBKSLWU6B/events.json","paper":"https://pith.science/paper/LSR2E3IS"},"agent_actions":{"view_html":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B","download_json":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B.json","view_paper":"https://pith.science/paper/LSR2E3IS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00647&json=true","fetch_graph":"https://pith.science/api/pith-number/LSR2E3ISP5LAY2TT3QBKSLWU6B/graph.json","fetch_events":"https://pith.science/api/pith-number/LSR2E3ISP5LAY2TT3QBKSLWU6B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B/action/storage_attestation","attest_author":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B/action/author_attestation","sign_citation":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B/action/citation_signature","submit_replication":"https://pith.science/pith/LSR2E3ISP5LAY2TT3QBKSLWU6B/action/replication_record"}},"created_at":"2026-07-05T07:36:03.391036+00:00","updated_at":"2026-07-05T07:36:03.391036+00:00"}