{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D3MBJ5WV662VKA4NSEWWPVZTO6","short_pith_number":"pith:D3MBJ5WV","schema_version":"1.0","canonical_sha256":"1ed814f6d5f7b555038d912d67d733779e3f7c49a2b072c60b6c85434e335fe7","source":{"kind":"arxiv","id":"2409.13989","version":1},"attestation_state":"computed","paper":{"title":"ChemEval: A Comprehensive Multi-Level Chemical Evaluation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","physics.chem-ph","q-bio.BM"],"primary_cat":"cs.CL","authors_text":"Defu Lian, Deguang Liu, Enhong Chen, Feiyang Xu, Guiquan Liu, Guoping Hu, Hao Wang, Huadong Liang, Jian Cui, Qi Liu, Rongyang Zhang, Shijin Wang, Xin Li, Xuesong He, Xuyang Zhi, Yi Li, Yuqing Huang, Zimu Liu","submitted_at":"2024-09-21T02:50:43Z","abstract_excerpt":"There is a growing interest in the role that LLMs play in chemistry which lead to an increased focus on the development of LLMs benchmarks tailored to chemical domains to assess the performance of LLMs across a spectrum of chemical tasks varying in type and complexity. However, existing benchmarks in this domain fail to adequately meet the specific requirements of chemical research professionals. To this end, we propose \\textbf{\\textit{ChemEval}}, which provides a comprehensive assessment of the capabilities of LLMs across a wide range of chemical domain tasks. Specifically, ChemEval identifie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13989","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-21T02:50:43Z","cross_cats_sorted":["cs.AI","cs.LG","physics.chem-ph","q-bio.BM"],"title_canon_sha256":"7ffe88ef2c1e3ef28406e4e0c4e310ed15f1e650c408256a4da4bff4df8ed852","abstract_canon_sha256":"d894e2cbc3300f2f3eaf1ac547df4ffe847a4776016cc9173af3fd77570c1e39"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:10.005417Z","signature_b64":"jXXSLfJITuHrbnY3WToPNJStGnT/+9/f1WKBcc7gnvHk4q5FuWNMcBG+DuvBYWNiJRcyeNStn7LPN29NXjc2Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ed814f6d5f7b555038d912d67d733779e3f7c49a2b072c60b6c85434e335fe7","last_reissued_at":"2026-07-05T09:10:10.004929Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:10.004929Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChemEval: A Comprehensive Multi-Level Chemical Evaluation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","physics.chem-ph","q-bio.BM"],"primary_cat":"cs.CL","authors_text":"Defu Lian, Deguang Liu, Enhong Chen, Feiyang Xu, Guiquan Liu, Guoping Hu, Hao Wang, Huadong Liang, Jian Cui, Qi Liu, Rongyang Zhang, Shijin Wang, Xin Li, Xuesong He, Xuyang Zhi, Yi Li, Yuqing Huang, Zimu Liu","submitted_at":"2024-09-21T02:50:43Z","abstract_excerpt":"There is a growing interest in the role that LLMs play in chemistry which lead to an increased focus on the development of LLMs benchmarks tailored to chemical domains to assess the performance of LLMs across a spectrum of chemical tasks varying in type and complexity. However, existing benchmarks in this domain fail to adequately meet the specific requirements of chemical research professionals. To this end, we propose \\textbf{\\textit{ChemEval}}, which provides a comprehensive assessment of the capabilities of LLMs across a wide range of chemical domain tasks. Specifically, ChemEval identifie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13989","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13989/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13989","created_at":"2026-07-05T09:10:10.004989+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13989v1","created_at":"2026-07-05T09:10:10.004989+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13989","created_at":"2026-07-05T09:10:10.004989+00:00"},{"alias_kind":"pith_short_12","alias_value":"D3MBJ5WV662V","created_at":"2026-07-05T09:10:10.004989+00:00"},{"alias_kind":"pith_short_16","alias_value":"D3MBJ5WV662VKA4N","created_at":"2026-07-05T09:10:10.004989+00:00"},{"alias_kind":"pith_short_8","alias_value":"D3MBJ5WV","created_at":"2026-07-05T09:10:10.004989+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03660","citing_title":"From Answers to States: Verifiable Process-Level Evaluation of Chemical Reasoning in Large Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21990","citing_title":"ChemDFM-R: A Chemical Reasoning LLM Enhanced with Atomized Chemical Knowledge","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18472","citing_title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","ref_index":102,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6","json":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6.json","graph_json":"https://pith.science/api/pith-number/D3MBJ5WV662VKA4NSEWWPVZTO6/graph.json","events_json":"https://pith.science/api/pith-number/D3MBJ5WV662VKA4NSEWWPVZTO6/events.json","paper":"https://pith.science/paper/D3MBJ5WV"},"agent_actions":{"view_html":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6","download_json":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6.json","view_paper":"https://pith.science/paper/D3MBJ5WV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13989&json=true","fetch_graph":"https://pith.science/api/pith-number/D3MBJ5WV662VKA4NSEWWPVZTO6/graph.json","fetch_events":"https://pith.science/api/pith-number/D3MBJ5WV662VKA4NSEWWPVZTO6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6/action/storage_attestation","attest_author":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6/action/author_attestation","sign_citation":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6/action/citation_signature","submit_replication":"https://pith.science/pith/D3MBJ5WV662VKA4NSEWWPVZTO6/action/replication_record"}},"created_at":"2026-07-05T09:10:10.004989+00:00","updated_at":"2026-07-05T09:10:10.004989+00:00"}