{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BSGFVEPILCAVIZPNVOWYTZKP7T","short_pith_number":"pith:BSGFVEPI","schema_version":"1.0","canonical_sha256":"0c8c5a91e858815465edabad89e54ffcfc9a526e6eada8e157eb4c87ce67bb27","source":{"kind":"arxiv","id":"2311.17092","version":1},"attestation_state":"computed","paper":{"title":"SEED-Bench-2: Benchmarking Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohao Li, Guangzhi Wang, Ruimao Zhang, Rui Wang, Ying Shan, Yixiao Ge, Yuying Ge","submitted_at":"2023-11-28T05:53:55Z","abstract_excerpt":"Multimodal large language models (MLLMs), building upon the foundation of powerful large language models (LLMs), have recently demonstrated exceptional capabilities in generating not only texts but also images given interleaved multimodal inputs (acting like a combination of GPT-4V and DALL-E 3). However, existing MLLM benchmarks remain limited to assessing only models' comprehension ability of single image-text inputs, failing to keep up with the strides made in MLLMs. A comprehensive benchmark is imperative for investigating the progress and uncovering the limitations of current MLLMs. In th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17092","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-28T05:53:55Z","cross_cats_sorted":[],"title_canon_sha256":"adee2096f55493672ce997f6f783d569343d666048fc8974b580d256c6fb7127","abstract_canon_sha256":"a727dfa5a5045f2b0dbd862d1733d47b575d827febb8f022e133c309fc920b20"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:57.363091Z","signature_b64":"veuoPHN6JM8BRTmD0xa974nfrprjGpOIlRJYav6cBwL4Ho9L980BDVZ/Kn8+BIeqXOOZY1fC0YiRxBPDp70+DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c8c5a91e858815465edabad89e54ffcfc9a526e6eada8e157eb4c87ce67bb27","last_reissued_at":"2026-07-05T07:17:57.362659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:57.362659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SEED-Bench-2: Benchmarking Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohao Li, Guangzhi Wang, Ruimao Zhang, Rui Wang, Ying Shan, Yixiao Ge, Yuying Ge","submitted_at":"2023-11-28T05:53:55Z","abstract_excerpt":"Multimodal large language models (MLLMs), building upon the foundation of powerful large language models (LLMs), have recently demonstrated exceptional capabilities in generating not only texts but also images given interleaved multimodal inputs (acting like a combination of GPT-4V and DALL-E 3). However, existing MLLM benchmarks remain limited to assessing only models' comprehension ability of single image-text inputs, failing to keep up with the strides made in MLLMs. A comprehensive benchmark is imperative for investigating the progress and uncovering the limitations of current MLLMs. In th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17092","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17092","created_at":"2026-07-05T07:17:57.362717+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17092v1","created_at":"2026-07-05T07:17:57.362717+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17092","created_at":"2026-07-05T07:17:57.362717+00:00"},{"alias_kind":"pith_short_12","alias_value":"BSGFVEPILCAV","created_at":"2026-07-05T07:17:57.362717+00:00"},{"alias_kind":"pith_short_16","alias_value":"BSGFVEPILCAVIZPN","created_at":"2026-07-05T07:17:57.362717+00:00"},{"alias_kind":"pith_short_8","alias_value":"BSGFVEPI","created_at":"2026-07-05T07:17:57.362717+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06964","citing_title":"End-to-End LLM Flight Planning with RAG-based Memory and Multi-modal Coach Agent","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07861","citing_title":"The Last Visible Pixel: Probing Fine-Scale Perception in Vision-Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06538","citing_title":"WorldBench: A Challenging and Visually Diverse Multimodal Reasoning Benchmark","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00535","citing_title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2405.19088","citing_title":"Cracking the Code of Juxtaposition: Can AI Models Understand the Humorous Contradictions","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23137","citing_title":"When 'YES' Meets 'BUT': Can Large Models Comprehend Contradictory Humor Through Comparative Reasoning?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2406.09411","citing_title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2404.12390","citing_title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27389","citing_title":"COHERENCE: Benchmarking Fine-Grained Image-Text Alignment in Interleaved Multimodal Contexts","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27389","citing_title":"COHERENCE: Benchmarking Fine-Grained Image-Text Alignment in Interleaved Multimodal Contexts","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T","json":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T.json","graph_json":"https://pith.science/api/pith-number/BSGFVEPILCAVIZPNVOWYTZKP7T/graph.json","events_json":"https://pith.science/api/pith-number/BSGFVEPILCAVIZPNVOWYTZKP7T/events.json","paper":"https://pith.science/paper/BSGFVEPI"},"agent_actions":{"view_html":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T","download_json":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T.json","view_paper":"https://pith.science/paper/BSGFVEPI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17092&json=true","fetch_graph":"https://pith.science/api/pith-number/BSGFVEPILCAVIZPNVOWYTZKP7T/graph.json","fetch_events":"https://pith.science/api/pith-number/BSGFVEPILCAVIZPNVOWYTZKP7T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T/action/storage_attestation","attest_author":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T/action/author_attestation","sign_citation":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T/action/citation_signature","submit_replication":"https://pith.science/pith/BSGFVEPILCAVIZPNVOWYTZKP7T/action/replication_record"}},"created_at":"2026-07-05T07:17:57.362717+00:00","updated_at":"2026-07-05T07:17:57.362717+00:00"}