{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YLGXDRM3KVZWIVRDHFLU5ONELE","short_pith_number":"pith:YLGXDRM3","schema_version":"1.0","canonical_sha256":"c2cd71c59b557364562339574eb9a4592371e7f689ac7f137f4ee6d5bb6afd84","source":{"kind":"arxiv","id":"2310.06083","version":1},"attestation_state":"computed","paper":{"title":"Transformers and Large Language Models for Chemistry and Drug Discovery","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["physics.chem-ph"],"primary_cat":"cs.LG","authors_text":"Andres M Bran, Philippe Schwaller","submitted_at":"2023-10-09T18:40:04Z","abstract_excerpt":"Language modeling has seen impressive progress over the last years, mainly prompted by the invention of the Transformer architecture, sparking a revolution in many fields of machine learning, with breakthroughs in chemistry and biology. In this chapter, we explore how analogies between chemical and natural language have inspired the use of Transformers to tackle important bottlenecks in the drug discovery process, such as retrosynthetic planning and chemical space exploration. The revolution started with models able to perform particular tasks with a single type of data, like linearised molecu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06083","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-09T18:40:04Z","cross_cats_sorted":["physics.chem-ph"],"title_canon_sha256":"539f6e0f65fd1d82630f1846829216ec5ee8a6186faa4c1ec24dcd017f6ab3bd","abstract_canon_sha256":"97b655ea871c287f2ac01a9471e15b1d6408cce8aec1d1e7e4ce320286c644ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:05.570356Z","signature_b64":"eX6l8b21PfQFQSLmnAth63w8XzvHBGSQfszphwIbQVb8kBsXjvg+bF3ofH5H6YRpwN+BOCHhmyePyqAAu8/MCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2cd71c59b557364562339574eb9a4592371e7f689ac7f137f4ee6d5bb6afd84","last_reissued_at":"2026-07-05T06:59:05.569917Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:05.569917Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers and Large Language Models for Chemistry and Drug Discovery","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["physics.chem-ph"],"primary_cat":"cs.LG","authors_text":"Andres M Bran, Philippe Schwaller","submitted_at":"2023-10-09T18:40:04Z","abstract_excerpt":"Language modeling has seen impressive progress over the last years, mainly prompted by the invention of the Transformer architecture, sparking a revolution in many fields of machine learning, with breakthroughs in chemistry and biology. In this chapter, we explore how analogies between chemical and natural language have inspired the use of Transformers to tackle important bottlenecks in the drug discovery process, such as retrosynthetic planning and chemical space exploration. The revolution started with models able to perform particular tasks with a single type of data, like linearised molecu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06083","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06083/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06083","created_at":"2026-07-05T06:59:05.569968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06083v1","created_at":"2026-07-05T06:59:05.569968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06083","created_at":"2026-07-05T06:59:05.569968+00:00"},{"alias_kind":"pith_short_12","alias_value":"YLGXDRM3KVZW","created_at":"2026-07-05T06:59:05.569968+00:00"},{"alias_kind":"pith_short_16","alias_value":"YLGXDRM3KVZWIVRD","created_at":"2026-07-05T06:59:05.569968+00:00"},{"alias_kind":"pith_short_8","alias_value":"YLGXDRM3","created_at":"2026-07-05T06:59:05.569968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.07237","citing_title":"DrugImproverGPT: A Large Language Model for Drug Optimization with Fine-Tuning via Structured Policy Optimization","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE","json":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE.json","graph_json":"https://pith.science/api/pith-number/YLGXDRM3KVZWIVRDHFLU5ONELE/graph.json","events_json":"https://pith.science/api/pith-number/YLGXDRM3KVZWIVRDHFLU5ONELE/events.json","paper":"https://pith.science/paper/YLGXDRM3"},"agent_actions":{"view_html":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE","download_json":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE.json","view_paper":"https://pith.science/paper/YLGXDRM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06083&json=true","fetch_graph":"https://pith.science/api/pith-number/YLGXDRM3KVZWIVRDHFLU5ONELE/graph.json","fetch_events":"https://pith.science/api/pith-number/YLGXDRM3KVZWIVRDHFLU5ONELE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE/action/storage_attestation","attest_author":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE/action/author_attestation","sign_citation":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE/action/citation_signature","submit_replication":"https://pith.science/pith/YLGXDRM3KVZWIVRDHFLU5ONELE/action/replication_record"}},"created_at":"2026-07-05T06:59:05.569968+00:00","updated_at":"2026-07-05T06:59:05.569968+00:00"}