{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HQ6LNWNU7RMA6FSF26HORW3GVB","short_pith_number":"pith:HQ6LNWNU","schema_version":"1.0","canonical_sha256":"3c3cb6d9b4fc580f1645d78ee8db66a8475ef66fd893c3e3a2636f5adba76a27","source":{"kind":"arxiv","id":"2407.04903","version":3},"attestation_state":"computed","paper":{"title":"MMSci: A Dataset for Graduate-Level Multi-Discipline Multimodal Scientific Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Byungju Lee, Hyeonjung Kim, Jin Hyuk Lim, Kyuri Choi, Linda Ruth Petzold, Ryan Hsieh, Stephen D. Wilson, Sungyoung Ji, Wanrong Zhu, William Yang Wang, Woosang Lim, Xianjun Yang, Xifeng Yan, Zekun Li","submitted_at":"2024-07-06T00:40:53Z","abstract_excerpt":"Scientific figure interpretation is a crucial capability for AI-driven scientific assistants built on advanced Large Vision Language Models. However, current datasets and benchmarks primarily focus on simple charts or other relatively straightforward figures from limited science domains. To address this gap, we present a comprehensive dataset compiled from peer-reviewed Nature Communications articles covering 72 scientific fields, encompassing complex visualizations such as schematic diagrams, microscopic images, and experimental data which require graduate-level expertise to interpret. We eva"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04903","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-06T00:40:53Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"8b90565966051a20ff2e12868f50b1a714d6491cd5b91cd726afb001852c8977","abstract_canon_sha256":"fd54616726ed427ea7781a7b9c4403fd62af8f85374dfa29168896a929e41153"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:09.593561Z","signature_b64":"xlUSKld/yypozrlYO3FUv6XHMkTUF1CDAoyxUwq84H+RXuBbcXRY0U1QJCcas0a+97LccxX0JNVLhPZ7hA95Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c3cb6d9b4fc580f1645d78ee8db66a8475ef66fd893c3e3a2636f5adba76a27","last_reissued_at":"2026-07-05T10:17:09.593020Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:09.593020Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMSci: A Dataset for Graduate-Level Multi-Discipline Multimodal Scientific Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Byungju Lee, Hyeonjung Kim, Jin Hyuk Lim, Kyuri Choi, Linda Ruth Petzold, Ryan Hsieh, Stephen D. Wilson, Sungyoung Ji, Wanrong Zhu, William Yang Wang, Woosang Lim, Xianjun Yang, Xifeng Yan, Zekun Li","submitted_at":"2024-07-06T00:40:53Z","abstract_excerpt":"Scientific figure interpretation is a crucial capability for AI-driven scientific assistants built on advanced Large Vision Language Models. However, current datasets and benchmarks primarily focus on simple charts or other relatively straightforward figures from limited science domains. To address this gap, we present a comprehensive dataset compiled from peer-reviewed Nature Communications articles covering 72 scientific fields, encompassing complex visualizations such as schematic diagrams, microscopic images, and experimental data which require graduate-level expertise to interpret. We eva"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04903","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04903/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04903","created_at":"2026-07-05T10:17:09.593079+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04903v3","created_at":"2026-07-05T10:17:09.593079+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04903","created_at":"2026-07-05T10:17:09.593079+00:00"},{"alias_kind":"pith_short_12","alias_value":"HQ6LNWNU7RMA","created_at":"2026-07-05T10:17:09.593079+00:00"},{"alias_kind":"pith_short_16","alias_value":"HQ6LNWNU7RMA6FSF","created_at":"2026-07-05T10:17:09.593079+00:00"},{"alias_kind":"pith_short_8","alias_value":"HQ6LNWNU","created_at":"2026-07-05T10:17:09.593079+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05222","citing_title":"A Multimodal Reasoning Typology for Grounding Chart-Image Coherence in Science Communication","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09871","citing_title":"SD-GRPO: Verifiable Segment Decomposition for Long-Form Vision-Language Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02212","citing_title":"SciGA: A Comprehensive Dataset for Designing Graphical Abstracts in Academic Papers","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13848","citing_title":"Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02813","citing_title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04172","citing_title":"GENFIG1: Visual Summaries of Scholarly Work as a Challenge for Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10302","citing_title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08211","citing_title":"SciFigDetect: A Benchmark for AI-Generated Scientific Figure Detection","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB","json":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB.json","graph_json":"https://pith.science/api/pith-number/HQ6LNWNU7RMA6FSF26HORW3GVB/graph.json","events_json":"https://pith.science/api/pith-number/HQ6LNWNU7RMA6FSF26HORW3GVB/events.json","paper":"https://pith.science/paper/HQ6LNWNU"},"agent_actions":{"view_html":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB","download_json":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB.json","view_paper":"https://pith.science/paper/HQ6LNWNU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04903&json=true","fetch_graph":"https://pith.science/api/pith-number/HQ6LNWNU7RMA6FSF26HORW3GVB/graph.json","fetch_events":"https://pith.science/api/pith-number/HQ6LNWNU7RMA6FSF26HORW3GVB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB/action/storage_attestation","attest_author":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB/action/author_attestation","sign_citation":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB/action/citation_signature","submit_replication":"https://pith.science/pith/HQ6LNWNU7RMA6FSF26HORW3GVB/action/replication_record"}},"created_at":"2026-07-05T10:17:09.593079+00:00","updated_at":"2026-07-05T10:17:09.593079+00:00"}