{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6ZQBCOZ5FVQFBF5A3M7JNP555P","short_pith_number":"pith:6ZQBCOZ5","schema_version":"1.0","canonical_sha256":"f660113b3d2d605097a0db3e96bfbdebc61a49acba85e54c193ae7ced9aa8d16","source":{"kind":"arxiv","id":"2312.15915","version":3},"attestation_state":"computed","paper":{"title":"ChartBench: A Benchmark for Complex Visual Reasoning in Charts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjin Xu, Chun Yuan, Jian Guo, Sinan Du, Yiyan Qi, Zhengzhuo Xu","submitted_at":"2023-12-26T07:20:55Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have shown impressive capabilities in image understanding and generation. However, current benchmarks fail to accurately evaluate the chart comprehension of MLLMs due to limited chart types and inappropriate metrics. To address this, we propose ChartBench, a comprehensive benchmark designed to assess chart comprehension and data reliability through complex visual reasoning. ChartBench includes 42 categories, 66.6k charts, and 600k question-answer pairs. Notably, many charts lack data point annotations, which requires MLLMs to derive values similar to hu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15915","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-12-26T07:20:55Z","cross_cats_sorted":[],"title_canon_sha256":"84d37367aec008a52036d031c1991435c47e7fa930b5cbd905e11ed8ae514c9e","abstract_canon_sha256":"27cd8a1b6c8208b70be529efd225cd1c57e2d6f13fd016abd4d031a033172740"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:09.326279Z","signature_b64":"XCRCDhuUXbRUpP/6YLjEV+PXvq0QcSKvvLXr6NRamvcxaNDD+d8+grLlBD64AeQb5yCtpfy88lAmI4j0k57ZBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f660113b3d2d605097a0db3e96bfbdebc61a49acba85e54c193ae7ced9aa8d16","last_reissued_at":"2026-07-05T08:34:09.325746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:09.325746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChartBench: A Benchmark for Complex Visual Reasoning in Charts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengjin Xu, Chun Yuan, Jian Guo, Sinan Du, Yiyan Qi, Zhengzhuo Xu","submitted_at":"2023-12-26T07:20:55Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have shown impressive capabilities in image understanding and generation. However, current benchmarks fail to accurately evaluate the chart comprehension of MLLMs due to limited chart types and inappropriate metrics. To address this, we propose ChartBench, a comprehensive benchmark designed to assess chart comprehension and data reliability through complex visual reasoning. ChartBench includes 42 categories, 66.6k charts, and 600k question-answer pairs. Notably, many charts lack data point annotations, which requires MLLMs to derive values similar to hu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15915","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15915","created_at":"2026-07-05T08:34:09.325810+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15915v3","created_at":"2026-07-05T08:34:09.325810+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15915","created_at":"2026-07-05T08:34:09.325810+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ZQBCOZ5FVQF","created_at":"2026-07-05T08:34:09.325810+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ZQBCOZ5FVQFBF5A","created_at":"2026-07-05T08:34:09.325810+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ZQBCOZ5","created_at":"2026-07-05T08:34:09.325810+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01558","citing_title":"Attention-guided Fine-tuning of Multimodal Large Language Models Improves Chain-of-Thought Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29808","citing_title":"Making Multimodal LLMs Reliable Chart Data Extractors: A Benchmark and Training Framework","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29446","citing_title":"CrystalXRD-Bench: Benchmarking Vision-Language Models for XRD Peak Indexing Across Diverse Crystalline Materials","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17962","citing_title":"FinDocMRE: A Benchmark for Document-Level Financial Multimodal Reasoning Evaluation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19173","citing_title":"CycleChart: A Unified Consistency-Based Learning Framework for Bidirectional Chart Understanding and Generation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13606","citing_title":"ChartVerse: Scaling Chart Reasoning via Reliable Programmatic Synthesis from Scratch","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13232","citing_title":"PlotChain: Deterministic Checkpointed Evaluation of Multimodal LLMs on Engineering Plot Reading","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02794","citing_title":"CharTool: Tool-Integrated Visual Reasoning for Chart Understanding","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01882","citing_title":"Chart-FR1: Visual Focus-Driven Fine-Grained Reasoning on Dense Charts","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18164","citing_title":"MM-JudgeBias: A Benchmark for Evaluating Compositional Biases in MLLM-as-a-Judge","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P","json":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P.json","graph_json":"https://pith.science/api/pith-number/6ZQBCOZ5FVQFBF5A3M7JNP555P/graph.json","events_json":"https://pith.science/api/pith-number/6ZQBCOZ5FVQFBF5A3M7JNP555P/events.json","paper":"https://pith.science/paper/6ZQBCOZ5"},"agent_actions":{"view_html":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P","download_json":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P.json","view_paper":"https://pith.science/paper/6ZQBCOZ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15915&json=true","fetch_graph":"https://pith.science/api/pith-number/6ZQBCOZ5FVQFBF5A3M7JNP555P/graph.json","fetch_events":"https://pith.science/api/pith-number/6ZQBCOZ5FVQFBF5A3M7JNP555P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P/action/storage_attestation","attest_author":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P/action/author_attestation","sign_citation":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P/action/citation_signature","submit_replication":"https://pith.science/pith/6ZQBCOZ5FVQFBF5A3M7JNP555P/action/replication_record"}},"created_at":"2026-07-05T08:34:09.325810+00:00","updated_at":"2026-07-05T08:34:09.325810+00:00"}