{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YZI5SG7SKTAK7MUGENG42LU456","short_pith_number":"pith:YZI5SG7S","schema_version":"1.0","canonical_sha256":"c651d91bf254c0afb286234dcd2e9cef89747cfdcf564ab18499f2a13be1546e","source":{"kind":"arxiv","id":"2404.16790","version":1},"attestation_state":"computed","paper":{"title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohao Li, Ruimao Zhang, Yi Chen, Ying Shan, Yixiao Ge, Yuying Ge","submitted_at":"2024-04-25T17:39:35Z","abstract_excerpt":"Comprehending text-rich visual content is paramount for the practical application of Multimodal Large Language Models (MLLMs), since text-rich scenarios are ubiquitous in the real world, which are characterized by the presence of extensive texts embedded within images. Recently, the advent of MLLMs with impressive versatility has raised the bar for what we can expect from MLLMs. However, their proficiency in text-rich scenarios has yet to be comprehensively and objectively assessed, since current MLLM benchmarks primarily focus on evaluating general visual comprehension. In this work, we intro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.16790","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-25T17:39:35Z","cross_cats_sorted":[],"title_canon_sha256":"e4da34efe1c0eedec31f3ee3fbc2f33854c5016bfaaa9c40475454deab015c62","abstract_canon_sha256":"d0abaff06dc3db3610664798a674d23daaf9dd2c7634bf509d09c57d383818f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:12:12.566758Z","signature_b64":"j2dsXiyXBW/R731foNyCjgoyy2hZ/3JzGFquZnykEw0mYb4xULkvHsPHI2RiB32NHZ2m+vxInBSrdAZbcyLjDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c651d91bf254c0afb286234dcd2e9cef89747cfdcf564ab18499f2a13be1546e","last_reissued_at":"2026-07-05T08:12:12.566292Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:12:12.566292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohao Li, Ruimao Zhang, Yi Chen, Ying Shan, Yixiao Ge, Yuying Ge","submitted_at":"2024-04-25T17:39:35Z","abstract_excerpt":"Comprehending text-rich visual content is paramount for the practical application of Multimodal Large Language Models (MLLMs), since text-rich scenarios are ubiquitous in the real world, which are characterized by the presence of extensive texts embedded within images. Recently, the advent of MLLMs with impressive versatility has raised the bar for what we can expect from MLLMs. However, their proficiency in text-rich scenarios has yet to be comprehensively and objectively assessed, since current MLLM benchmarks primarily focus on evaluating general visual comprehension. In this work, we intro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.16790","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.16790/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.16790","created_at":"2026-07-05T08:12:12.566366+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.16790v1","created_at":"2026-07-05T08:12:12.566366+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.16790","created_at":"2026-07-05T08:12:12.566366+00:00"},{"alias_kind":"pith_short_12","alias_value":"YZI5SG7SKTAK","created_at":"2026-07-05T08:12:12.566366+00:00"},{"alias_kind":"pith_short_16","alias_value":"YZI5SG7SKTAK7MUG","created_at":"2026-07-05T08:12:12.566366+00:00"},{"alias_kind":"pith_short_8","alias_value":"YZI5SG7S","created_at":"2026-07-05T08:12:12.566366+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20328","citing_title":"HyLaR: Hybrid Latent Reasoning with Decoupled Policy Optimization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24602","citing_title":"ViTexQA: A Multi-Frame Temporal Perception Dataset for Video Text Question Answering","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09393","citing_title":"CapRL++: Unified Reinforcement Learning with Verifiable Rewards for Dense Image and Video Captioning","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07861","citing_title":"The Last Visible Pixel: Probing Fine-Scale Perception in Vision-Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00465","citing_title":"StochasT: Learning with Stochastic Turn Depth for Visual Instruction Tuning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12960","citing_title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14998","citing_title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2601.06803","citing_title":"Forest Before Trees: Latent Superposition for Efficient Visual Reasoning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05271","citing_title":"DeepEyesV2: Toward Agentic Multimodal Model","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12960","citing_title":"DiM\\textsuperscript{3}: Bridging Multilingual and Multimodal Models via Direction- and Magnitude-Aware Merging","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20328","citing_title":"HyLaR: Hybrid Latent Reasoning with Decoupled Policy Optimization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08545","citing_title":"Act Wisely: Cultivating Meta-Cognitive Tool Use in Agentic Multimodal Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18265","citing_title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456","json":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456.json","graph_json":"https://pith.science/api/pith-number/YZI5SG7SKTAK7MUGENG42LU456/graph.json","events_json":"https://pith.science/api/pith-number/YZI5SG7SKTAK7MUGENG42LU456/events.json","paper":"https://pith.science/paper/YZI5SG7S"},"agent_actions":{"view_html":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456","download_json":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456.json","view_paper":"https://pith.science/paper/YZI5SG7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.16790&json=true","fetch_graph":"https://pith.science/api/pith-number/YZI5SG7SKTAK7MUGENG42LU456/graph.json","fetch_events":"https://pith.science/api/pith-number/YZI5SG7SKTAK7MUGENG42LU456/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456/action/storage_attestation","attest_author":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456/action/author_attestation","sign_citation":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456/action/citation_signature","submit_replication":"https://pith.science/pith/YZI5SG7SKTAK7MUGENG42LU456/action/replication_record"}},"created_at":"2026-07-05T08:12:12.566366+00:00","updated_at":"2026-07-05T08:12:12.566366+00:00"}