{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7EA4AI3NCZQ5YAVNTOUNXQX7G6","short_pith_number":"pith:7EA4AI3N","schema_version":"1.0","canonical_sha256":"f901c0236d1661dc02ad9ba8dbc2ff37b385fc6ad1dde522fb2314c34ac6ece3","source":{"kind":"arxiv","id":"2409.13730","version":2},"attestation_state":"computed","paper":{"title":"VisScience: An Extensive Benchmark for Evaluating K12 Educational Multi-modal Scientific Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Bin Xu, Jie Tang, Jinhao Chen, Weihan Wang, Zhengxiao Du, Zhen Yang, Zhihuan Jiang","submitted_at":"2024-09-10T01:20:26Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have demonstrated promising capabilities across various tasks by integrating textual and visual information to achieve visual understanding in complex scenarios. Despite the availability of several benchmarks aims to evaluating MLLMs in tasks from visual question answering to complex problem-solving, most focus predominantly on mathematics or general visual understanding tasks. This reveals a critical gap in current benchmarks, which often overlook the inclusion of other key scientific disciplines such as physics and chemistry. To address this gap, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13730","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-09-10T01:20:26Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ee7c82bd23145d06fcea8b4ff95ef65887fc7de1125dc5397537af72a5c03cb3","abstract_canon_sha256":"5b49b1332effe8b8f0e61868653f233c375380e80097b88a2a67f16f74c9f6d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:43:09.708828Z","signature_b64":"BdS6dW2ZbuuXKIqrA/0wK90RXgl3RUIFGXQUAJCsvPjHB256cmdhgE6jOfk4/eRr8nDBATJ2nQt4qjEELCybDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f901c0236d1661dc02ad9ba8dbc2ff37b385fc6ad1dde522fb2314c34ac6ece3","last_reissued_at":"2026-07-05T09:43:09.708283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:43:09.708283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisScience: An Extensive Benchmark for Evaluating K12 Educational Multi-modal Scientific Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Bin Xu, Jie Tang, Jinhao Chen, Weihan Wang, Zhengxiao Du, Zhen Yang, Zhihuan Jiang","submitted_at":"2024-09-10T01:20:26Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have demonstrated promising capabilities across various tasks by integrating textual and visual information to achieve visual understanding in complex scenarios. Despite the availability of several benchmarks aims to evaluating MLLMs in tasks from visual question answering to complex problem-solving, most focus predominantly on mathematics or general visual understanding tasks. This reveals a critical gap in current benchmarks, which often overlook the inclusion of other key scientific disciplines such as physics and chemistry. To address this gap, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13730","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13730/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13730","created_at":"2026-07-05T09:43:09.708358+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13730v2","created_at":"2026-07-05T09:43:09.708358+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13730","created_at":"2026-07-05T09:43:09.708358+00:00"},{"alias_kind":"pith_short_12","alias_value":"7EA4AI3NCZQ5","created_at":"2026-07-05T09:43:09.708358+00:00"},{"alias_kind":"pith_short_16","alias_value":"7EA4AI3NCZQ5YAVN","created_at":"2026-07-05T09:43:09.708358+00:00"},{"alias_kind":"pith_short_8","alias_value":"7EA4AI3N","created_at":"2026-07-05T09:43:09.708358+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05966","citing_title":"Causal Scaffolding for Physical Reasoning: A Benchmark for Causally-Informed Physical World Understanding in VLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30673","citing_title":"TeachObs: A Human-Validated Benchmark for Multimodal Teaching Observation and Model Evaluation","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6","json":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6.json","graph_json":"https://pith.science/api/pith-number/7EA4AI3NCZQ5YAVNTOUNXQX7G6/graph.json","events_json":"https://pith.science/api/pith-number/7EA4AI3NCZQ5YAVNTOUNXQX7G6/events.json","paper":"https://pith.science/paper/7EA4AI3N"},"agent_actions":{"view_html":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6","download_json":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6.json","view_paper":"https://pith.science/paper/7EA4AI3N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13730&json=true","fetch_graph":"https://pith.science/api/pith-number/7EA4AI3NCZQ5YAVNTOUNXQX7G6/graph.json","fetch_events":"https://pith.science/api/pith-number/7EA4AI3NCZQ5YAVNTOUNXQX7G6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6/action/storage_attestation","attest_author":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6/action/author_attestation","sign_citation":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6/action/citation_signature","submit_replication":"https://pith.science/pith/7EA4AI3NCZQ5YAVNTOUNXQX7G6/action/replication_record"}},"created_at":"2026-07-05T09:43:09.708358+00:00","updated_at":"2026-07-05T09:43:09.708358+00:00"}