{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X5GEEHNO4JSZO2X4CAQYWYYCUF","short_pith_number":"pith:X5GEEHNO","schema_version":"1.0","canonical_sha256":"bf4c421daee265976afc10218b6302a16e001e42c6861e6149eaaca81c42ca6c","source":{"kind":"arxiv","id":"2402.09391","version":4},"attestation_state":"computed","paper":{"title":"LlaSMol: Advancing Large Language Models for Chemistry with a Large-Scale, Comprehensive, High-Quality Instruction Tuning Dataset","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CE","cs.CL"],"primary_cat":"cs.AI","authors_text":"Botao Yu, Frazier N. Baker, Huan Sun, Xia Ning, Ziqi Chen","submitted_at":"2024-02-14T18:42:25Z","abstract_excerpt":"Chemistry plays a crucial role in many domains, such as drug discovery and material science. While large language models (LLMs) such as GPT-4 exhibit remarkable capabilities on natural language processing tasks, existing research indicates that their performance on chemistry tasks is discouragingly low. In this paper, however, we demonstrate that our developed LLMs can achieve very strong results on a comprehensive set of chemistry tasks, outperforming the most advanced GPT-4 and Claude 3 Opus by a substantial margin. To accomplish this, we propose SMolInstruct, a large-scale, comprehensive, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.09391","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-02-14T18:42:25Z","cross_cats_sorted":["cs.CE","cs.CL"],"title_canon_sha256":"0530136ebea5629a56f07fda620ff7ea7fe6cbc887f39115e4e020ab5b0d8c81","abstract_canon_sha256":"0ce0a3fa162774e4e74e31cfecd65a3303a570e46913a178a51cde7b4dcbd73d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:07.152211Z","signature_b64":"buoweowf5RLk1Ya6+KOErZDBugIkHgjcPLQt7fTek0rWJrHtAPtYfhHLUr7doD9wuGSHzj5czDQM7u0MDNkzDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf4c421daee265976afc10218b6302a16e001e42c6861e6149eaaca81c42ca6c","last_reissued_at":"2026-07-05T08:54:07.151808Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:07.151808Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LlaSMol: Advancing Large Language Models for Chemistry with a Large-Scale, Comprehensive, High-Quality Instruction Tuning Dataset","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CE","cs.CL"],"primary_cat":"cs.AI","authors_text":"Botao Yu, Frazier N. Baker, Huan Sun, Xia Ning, Ziqi Chen","submitted_at":"2024-02-14T18:42:25Z","abstract_excerpt":"Chemistry plays a crucial role in many domains, such as drug discovery and material science. While large language models (LLMs) such as GPT-4 exhibit remarkable capabilities on natural language processing tasks, existing research indicates that their performance on chemistry tasks is discouragingly low. In this paper, however, we demonstrate that our developed LLMs can achieve very strong results on a comprehensive set of chemistry tasks, outperforming the most advanced GPT-4 and Claude 3 Opus by a substantial margin. To accomplish this, we propose SMolInstruct, a large-scale, comprehensive, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09391","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09391/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.09391","created_at":"2026-07-05T08:54:07.151862+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.09391v4","created_at":"2026-07-05T08:54:07.151862+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09391","created_at":"2026-07-05T08:54:07.151862+00:00"},{"alias_kind":"pith_short_12","alias_value":"X5GEEHNO4JSZ","created_at":"2026-07-05T08:54:07.151862+00:00"},{"alias_kind":"pith_short_16","alias_value":"X5GEEHNO4JSZO2X4","created_at":"2026-07-05T08:54:07.151862+00:00"},{"alias_kind":"pith_short_8","alias_value":"X5GEEHNO","created_at":"2026-07-05T08:54:07.151862+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13477","citing_title":"SupraBench: A Benchmark for Supramolecular Chemistry","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05693","citing_title":"MolE-RAG: Molecular Structure-Enhanced Retrieval-Augmented Generation for Chemistry","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19276","citing_title":"OpenCompass: A Universal Evaluation Platform for Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28862","citing_title":"Molecular Lead Optimization via Agentic Tool Planning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02231","citing_title":"SmileyLlama: Modifying Large Language Models for Directed Chemical Space Exploration","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2409.06080","citing_title":"Regression with Large Language Models for Materials and Molecular Property Prediction","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22287","citing_title":"SciCore-Mol: Augmenting Large Language Models with Pluggable Molecular Cognition Modules","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20740","citing_title":"Distribution-Aware Reward: Reinforcement Learning over Predictive Distributions for LLM Regression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19276","citing_title":"OpenCompass: A Universal Evaluation Platform for Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21990","citing_title":"ChemDFM-R: A Chemical Reasoning LLM Enhanced with Atomized Chemical Knowledge","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03361","citing_title":"The limits of bio-molecular modeling with large language models : a cross-scale evaluation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27351","citing_title":"Heterogeneous Scientific Foundation Model Collaboration","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10230","citing_title":"FORGE: Fragment-Oriented Ranking and Generation for Context-Aware Molecular Optimization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02745","citing_title":"Bolek: A Multimodal Language Model for Molecular Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07251","citing_title":"Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04403","citing_title":"MolDA: Molecular Understanding and Generation via Large Language Diffusion Model","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF","json":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF.json","graph_json":"https://pith.science/api/pith-number/X5GEEHNO4JSZO2X4CAQYWYYCUF/graph.json","events_json":"https://pith.science/api/pith-number/X5GEEHNO4JSZO2X4CAQYWYYCUF/events.json","paper":"https://pith.science/paper/X5GEEHNO"},"agent_actions":{"view_html":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF","download_json":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF.json","view_paper":"https://pith.science/paper/X5GEEHNO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.09391&json=true","fetch_graph":"https://pith.science/api/pith-number/X5GEEHNO4JSZO2X4CAQYWYYCUF/graph.json","fetch_events":"https://pith.science/api/pith-number/X5GEEHNO4JSZO2X4CAQYWYYCUF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF/action/storage_attestation","attest_author":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF/action/author_attestation","sign_citation":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF/action/citation_signature","submit_replication":"https://pith.science/pith/X5GEEHNO4JSZO2X4CAQYWYYCUF/action/replication_record"}},"created_at":"2026-07-05T08:54:07.151862+00:00","updated_at":"2026-07-05T08:54:07.151862+00:00"}