{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6Y2ZOC666FURG5QT4BQLKY2VES","short_pith_number":"pith:6Y2ZOC66","schema_version":"1.0","canonical_sha256":"f635970bdef169137613e060b5635524a1cb3b8e0e3425db078b14e600ff1477","source":{"kind":"arxiv","id":"2402.00024","version":3},"attestation_state":"computed","paper":{"title":"Can Large Language Models Understand Molecules?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"q-bio.BM","authors_text":"Alan Bui, Ali Forooghi, Alioune Ngom, Jianguo Lu, Shaghayegh Sadeghi","submitted_at":"2024-01-05T18:31:34Z","abstract_excerpt":"Purpose: Large Language Models (LLMs) like GPT (Generative Pre-trained Transformer) from OpenAI and LLaMA (Large Language Model Meta AI) from Meta AI are increasingly recognized for their potential in the field of cheminformatics, particularly in understanding Simplified Molecular Input Line Entry System (SMILES), a standard method for representing chemical structures. These LLMs also have the ability to decode SMILES strings into vector representations.\n  Method: We investigate the performance of GPT and LLaMA compared to pre-trained models on SMILES in embedding SMILES strings on downstream "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00024","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"q-bio.BM","submitted_at":"2024-01-05T18:31:34Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"062f97810fc6b9bc4a2e067f998c42b61376ea50acac3592cf9b09ef92d2846d","abstract_canon_sha256":"bfcfc86793af72c79462bf33caf39d629837bd5834e01e80b518d7602d16096f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:21:12.284528Z","signature_b64":"XFS+33yHgssgqcW+dRc0SCm4jtYnNNWgSS8eic8nqDX8M7Z6AOYemS/fa5OaR4SSDFLjbjI/MTJHnQlREBpKAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f635970bdef169137613e060b5635524a1cb3b8e0e3425db078b14e600ff1477","last_reissued_at":"2026-07-05T08:21:12.284042Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:21:12.284042Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Understand Molecules?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"q-bio.BM","authors_text":"Alan Bui, Ali Forooghi, Alioune Ngom, Jianguo Lu, Shaghayegh Sadeghi","submitted_at":"2024-01-05T18:31:34Z","abstract_excerpt":"Purpose: Large Language Models (LLMs) like GPT (Generative Pre-trained Transformer) from OpenAI and LLaMA (Large Language Model Meta AI) from Meta AI are increasingly recognized for their potential in the field of cheminformatics, particularly in understanding Simplified Molecular Input Line Entry System (SMILES), a standard method for representing chemical structures. These LLMs also have the ability to decode SMILES strings into vector representations.\n  Method: We investigate the performance of GPT and LLaMA compared to pre-trained models on SMILES in embedding SMILES strings on downstream "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00024","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00024/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00024","created_at":"2026-07-05T08:21:12.284101+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00024v3","created_at":"2026-07-05T08:21:12.284101+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00024","created_at":"2026-07-05T08:21:12.284101+00:00"},{"alias_kind":"pith_short_12","alias_value":"6Y2ZOC666FUR","created_at":"2026-07-05T08:21:12.284101+00:00"},{"alias_kind":"pith_short_16","alias_value":"6Y2ZOC666FURG5QT","created_at":"2026-07-05T08:21:12.284101+00:00"},{"alias_kind":"pith_short_8","alias_value":"6Y2ZOC66","created_at":"2026-07-05T08:21:12.284101+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18497","citing_title":"Leveraging neural network interatomic potentials for a foundation model of chemistry","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES","json":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES.json","graph_json":"https://pith.science/api/pith-number/6Y2ZOC666FURG5QT4BQLKY2VES/graph.json","events_json":"https://pith.science/api/pith-number/6Y2ZOC666FURG5QT4BQLKY2VES/events.json","paper":"https://pith.science/paper/6Y2ZOC66"},"agent_actions":{"view_html":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES","download_json":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES.json","view_paper":"https://pith.science/paper/6Y2ZOC66","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00024&json=true","fetch_graph":"https://pith.science/api/pith-number/6Y2ZOC666FURG5QT4BQLKY2VES/graph.json","fetch_events":"https://pith.science/api/pith-number/6Y2ZOC666FURG5QT4BQLKY2VES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES/action/storage_attestation","attest_author":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES/action/author_attestation","sign_citation":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES/action/citation_signature","submit_replication":"https://pith.science/pith/6Y2ZOC666FURG5QT4BQLKY2VES/action/replication_record"}},"created_at":"2026-07-05T08:21:12.284101+00:00","updated_at":"2026-07-05T08:21:12.284101+00:00"}