{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O73OWDLSK3HQ5WJAI7YLYAZP3I","short_pith_number":"pith:O73OWDLS","schema_version":"1.0","canonical_sha256":"77f6eb0d7256cf0ed92047f0bc032fda0284f620c0bbfe3296b60ae5d6891367","source":{"kind":"arxiv","id":"2505.18319","version":1},"attestation_state":"computed","paper":{"title":"Seeing Beyond Words: MatVQA for Challenging Visual-Scientific Reasoning in Materials Science","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CE","authors_text":"Amirreza Ataei, Bang Liu, Farshid Effaty, Huan Zhang, Sifan Wu, Yizhan Li","submitted_at":"2025-05-23T19:26:47Z","abstract_excerpt":"The emergence of Multimodal Large Language Models (MLLMs) that integrate vision and language modalities has unlocked new potentials for scientific reasoning, outperforming prior benchmarks in both natural language and coding domains. Current materials science evaluation datasets such as MaScQA and SciQA remain largely text-based and fail to capture the visual and research-level analytic complexity required in materials discovery and design. We introduce MatVQA, a scalable benchmark specifically designed to address this gap. Generated via an automated pipeline, MArxivAgent, from recent material"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18319","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CE","submitted_at":"2025-05-23T19:26:47Z","cross_cats_sorted":[],"title_canon_sha256":"8ddc4d49b4508fe90537ca2f85ad9560ba744f1c2f232abdec33a70d3c00dc5e","abstract_canon_sha256":"ff17f08af93dc9f7717162b6596d7f5d7b027bee316a7fa596986d8e62be8227"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:44.644043Z","signature_b64":"OOUu+sPS1erDc9e8jxnQhMYFASXN5nBn+WhG8BOeGT0XSRurBVX9zRXF9vDacaWBk3HM1rVBRKQY6jr1F82nDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77f6eb0d7256cf0ed92047f0bc032fda0284f620c0bbfe3296b60ae5d6891367","last_reissued_at":"2026-07-05T11:08:44.643506Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:44.643506Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Seeing Beyond Words: MatVQA for Challenging Visual-Scientific Reasoning in Materials Science","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CE","authors_text":"Amirreza Ataei, Bang Liu, Farshid Effaty, Huan Zhang, Sifan Wu, Yizhan Li","submitted_at":"2025-05-23T19:26:47Z","abstract_excerpt":"The emergence of Multimodal Large Language Models (MLLMs) that integrate vision and language modalities has unlocked new potentials for scientific reasoning, outperforming prior benchmarks in both natural language and coding domains. Current materials science evaluation datasets such as MaScQA and SciQA remain largely text-based and fail to capture the visual and research-level analytic complexity required in materials discovery and design. We introduce MatVQA, a scalable benchmark specifically designed to address this gap. Generated via an automated pipeline, MArxivAgent, from recent material"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18319","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18319","created_at":"2026-07-05T11:08:44.643577+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18319v1","created_at":"2026-07-05T11:08:44.643577+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18319","created_at":"2026-07-05T11:08:44.643577+00:00"},{"alias_kind":"pith_short_12","alias_value":"O73OWDLSK3HQ","created_at":"2026-07-05T11:08:44.643577+00:00"},{"alias_kind":"pith_short_16","alias_value":"O73OWDLSK3HQ5WJA","created_at":"2026-07-05T11:08:44.643577+00:00"},{"alias_kind":"pith_short_8","alias_value":"O73OWDLS","created_at":"2026-07-05T11:08:44.643577+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29833","citing_title":"OmniMatBench: A Human-Calibrated Multimodal Reasoning Benchmark Across 19 Materials Science Subfields","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I","json":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I.json","graph_json":"https://pith.science/api/pith-number/O73OWDLSK3HQ5WJAI7YLYAZP3I/graph.json","events_json":"https://pith.science/api/pith-number/O73OWDLSK3HQ5WJAI7YLYAZP3I/events.json","paper":"https://pith.science/paper/O73OWDLS"},"agent_actions":{"view_html":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I","download_json":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I.json","view_paper":"https://pith.science/paper/O73OWDLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18319&json=true","fetch_graph":"https://pith.science/api/pith-number/O73OWDLSK3HQ5WJAI7YLYAZP3I/graph.json","fetch_events":"https://pith.science/api/pith-number/O73OWDLSK3HQ5WJAI7YLYAZP3I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I/action/storage_attestation","attest_author":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I/action/author_attestation","sign_citation":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I/action/citation_signature","submit_replication":"https://pith.science/pith/O73OWDLSK3HQ5WJAI7YLYAZP3I/action/replication_record"}},"created_at":"2026-07-05T11:08:44.643577+00:00","updated_at":"2026-07-05T11:08:44.643577+00:00"}