{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:LT5IFTO74TAI4CDVKES4R6V4R7","short_pith_number":"pith:LT5IFTO7","schema_version":"1.0","canonical_sha256":"5cfa82cddfe4c08e08755125c8fabc8fdc9519f42038b9651d5c934fa7f45581","source":{"kind":"arxiv","id":"2608.06931","version":1},"attestation_state":"computed","paper":{"title":"Science Edge Evaluation: SEE the Missing Step Toward Real Scientific Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bing Zhao, Chen Zhao, Hu Wei, Jiajia Li, Jiaxin Li, Jinghang Wang, Jinxin Wang, Junchao Li, Kewei Sun, Lin Qu, Qile Jin, Qingteng Chen, Renquan Lv, Ruodan Chen, Shuai Bai, Shuang Wu, Taolin Han, Wai Yuet Chiu, Weiqi Zhai, Yifei Zhang, Yuchen Zhang, Yuhao Zhou, Yun Wu, Zhaohai Li, Zhibo Yang","submitted_at":"2026-08-07T08:06:31Z","abstract_excerpt":"Large language models (LLMs) are increasingly involved in scientific discovery, yet it remains unclear whether they can support complex real laboratory science. Here we introduce Science Edge Evaluation (SEE), a multimodal benchmark of expert-curated questions grounded in peer-reviewed literature and experimental practice in chemistry, biology, and materials science. Evaluation of 19 multimodal large language models (MLLMs) shows that even the best-performing model reaches only 48.7% accuracy. Moreover, general-purpose models outperform science-specialized models on average. In the visual-agen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.06931","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T08:06:31Z","cross_cats_sorted":[],"title_canon_sha256":"9e17313a767675bf62a9d7dfe90f9f4cf4b6f2e9152945070f6b7eb6469e936c","abstract_canon_sha256":"55f4b183ef097c9f4013f61775b386c736227f54890c9d5928b9be5abb76bc70"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-10T01:12:33.770970Z","signature_b64":"TyX+sIsPG0SFj0j63vDs20O9DF36sBE6/FxgcPPK5HpL58s66QJHtaB5QBa77xIoyaSUUW9UlD0O/vk7qULXCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5cfa82cddfe4c08e08755125c8fabc8fdc9519f42038b9651d5c934fa7f45581","last_reissued_at":"2026-08-10T01:12:33.768561Z","signature_status":"signed_v1","first_computed_at":"2026-08-10T01:12:33.768561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Science Edge Evaluation: SEE the Missing Step Toward Real Scientific Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bing Zhao, Chen Zhao, Hu Wei, Jiajia Li, Jiaxin Li, Jinghang Wang, Jinxin Wang, Junchao Li, Kewei Sun, Lin Qu, Qile Jin, Qingteng Chen, Renquan Lv, Ruodan Chen, Shuai Bai, Shuang Wu, Taolin Han, Wai Yuet Chiu, Weiqi Zhai, Yifei Zhang, Yuchen Zhang, Yuhao Zhou, Yun Wu, Zhaohai Li, Zhibo Yang","submitted_at":"2026-08-07T08:06:31Z","abstract_excerpt":"Large language models (LLMs) are increasingly involved in scientific discovery, yet it remains unclear whether they can support complex real laboratory science. Here we introduce Science Edge Evaluation (SEE), a multimodal benchmark of expert-curated questions grounded in peer-reviewed literature and experimental practice in chemistry, biology, and materials science. Evaluation of 19 multimodal large language models (MLLMs) shows that even the best-performing model reaches only 48.7% accuracy. Moreover, general-purpose models outperform science-specialized models on average. In the visual-agen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.06931","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.06931/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.06931","created_at":"2026-08-10T01:12:33.769454+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.06931v1","created_at":"2026-08-10T01:12:33.769454+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.06931","created_at":"2026-08-10T01:12:33.769454+00:00"},{"alias_kind":"pith_short_12","alias_value":"LT5IFTO74TAI","created_at":"2026-08-10T01:12:33.769454+00:00"},{"alias_kind":"pith_short_16","alias_value":"LT5IFTO74TAI4CDV","created_at":"2026-08-10T01:12:33.769454+00:00"},{"alias_kind":"pith_short_8","alias_value":"LT5IFTO7","created_at":"2026-08-10T01:12:33.769454+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7","json":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7.json","graph_json":"https://pith.science/api/pith-number/LT5IFTO74TAI4CDVKES4R6V4R7/graph.json","events_json":"https://pith.science/api/pith-number/LT5IFTO74TAI4CDVKES4R6V4R7/events.json","paper":"https://pith.science/paper/LT5IFTO7"},"agent_actions":{"view_html":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7","download_json":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7.json","view_paper":"https://pith.science/paper/LT5IFTO7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.06931&json=true","fetch_graph":"https://pith.science/api/pith-number/LT5IFTO74TAI4CDVKES4R6V4R7/graph.json","fetch_events":"https://pith.science/api/pith-number/LT5IFTO74TAI4CDVKES4R6V4R7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7/action/storage_attestation","attest_author":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7/action/author_attestation","sign_citation":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7/action/citation_signature","submit_replication":"https://pith.science/pith/LT5IFTO74TAI4CDVKES4R6V4R7/action/replication_record"}},"created_at":"2026-08-10T01:12:33.769454+00:00","updated_at":"2026-08-10T01:12:33.769454+00:00"}