{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:XFETXKGV3HJNNVROPCC7EQI2KJ","short_pith_number":"pith:XFETXKGV","schema_version":"1.0","canonical_sha256":"b9493ba8d5d9d2d6d62e7885f2411a527e23f35dca2b8ab00d35a34baa300455","source":{"kind":"arxiv","id":"2209.01712","version":1},"attestation_state":"computed","paper":{"title":"ChemBERTa-2: Towards Chemical Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.BM"],"primary_cat":"cs.LG","authors_text":"Bharath Ramsundar, Elana Simon, Gabriel Grand, Seyone Chithrananda, Walid Ahmad","submitted_at":"2022-09-05T00:31:12Z","abstract_excerpt":"Large pretrained models such as GPT-3 have had tremendous impact on modern natural language processing by leveraging self-supervised learning to learn salient representations that can be used to readily finetune on a wide variety of downstream tasks. We investigate the possibility of transferring such advances to molecular machine learning by building a chemical foundation model, ChemBERTa-2, using the language of SMILES. While labeled data for molecular prediction tasks is typically scarce, libraries of SMILES strings are readily available. In this work, we build upon ChemBERTa by optimizing "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.01712","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-09-05T00:31:12Z","cross_cats_sorted":["cs.AI","q-bio.BM"],"title_canon_sha256":"58fe0fa8821086ae5433100f4e68457cffafef45fdf7c4679d13397bdb8f759b","abstract_canon_sha256":"a9d0fb5cebe2cbe10e1c879ba8552180b46a5f03ff60eaa42850aa7e4ba005e1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:54:33.035358Z","signature_b64":"mypyjDzE8zkkuEDcif6EHgIYprVuoVTWVFK9DBz9yUuhmGT2Y4nyfEbvFRImUmPIbHXhti6eSmaZdUzRa8J8BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9493ba8d5d9d2d6d62e7885f2411a527e23f35dca2b8ab00d35a34baa300455","last_reissued_at":"2026-07-05T04:54:33.034923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:54:33.034923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChemBERTa-2: Towards Chemical Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.BM"],"primary_cat":"cs.LG","authors_text":"Bharath Ramsundar, Elana Simon, Gabriel Grand, Seyone Chithrananda, Walid Ahmad","submitted_at":"2022-09-05T00:31:12Z","abstract_excerpt":"Large pretrained models such as GPT-3 have had tremendous impact on modern natural language processing by leveraging self-supervised learning to learn salient representations that can be used to readily finetune on a wide variety of downstream tasks. We investigate the possibility of transferring such advances to molecular machine learning by building a chemical foundation model, ChemBERTa-2, using the language of SMILES. While labeled data for molecular prediction tasks is typically scarce, libraries of SMILES strings are readily available. In this work, we build upon ChemBERTa by optimizing "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.01712","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.01712/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.01712","created_at":"2026-07-05T04:54:33.034982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.01712v1","created_at":"2026-07-05T04:54:33.034982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.01712","created_at":"2026-07-05T04:54:33.034982+00:00"},{"alias_kind":"pith_short_12","alias_value":"XFETXKGV3HJN","created_at":"2026-07-05T04:54:33.034982+00:00"},{"alias_kind":"pith_short_16","alias_value":"XFETXKGV3HJNNVRO","created_at":"2026-07-05T04:54:33.034982+00:00"},{"alias_kind":"pith_short_8","alias_value":"XFETXKGV","created_at":"2026-07-05T04:54:33.034982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23443","citing_title":"What Does a Chemical Language Model Know About Molecules?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12639","citing_title":"The Metric Picks the Winner: Evaluation Choice Flips Model Rankings for Drug-Response Prediction in Unseen Chemistry","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28843","citing_title":"The Biosecurity Blind Spot: Systematic Dual-use Detection in Open Science Infrastructure","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22287","citing_title":"SciCore-Mol: Augmenting Large Language Models with Pluggable Molecular Cognition Modules","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08683","citing_title":"Machine learning for smell: Ordinal odor strength prediction of molecular perfumery components","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19752","citing_title":"MSAlign: Aligning Molecule and Mass Spectra Foundation Models for Metabolite Identification","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26498","citing_title":"Do Larger Models Really Win in Drug Discovery? A Benchmark Assessment of Model Scaling in AI-Driven Molecular Property and Activity Prediction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06484","citing_title":"Thermodynamically consistent machine learning model for excess Gibbs energy","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07731","citing_title":"oMeBench: Towards Robust Benchmarking of LLMs in Organic Mechanism Elucidation and Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18900","citing_title":"Foundation Models for Discovery and Exploration in Chemical Space","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13262","citing_title":"Chem-GMNet: A Sphere-Native Geometric Transformer for Molecular Property Prediction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26498","citing_title":"Do Larger Models Really Win in Drug Discovery? A Benchmark Assessment of Model Scaling in AI-Driven Molecular Property and Activity Prediction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02745","citing_title":"Bolek: A Multimodal Language Model for Molecular Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05622","citing_title":"CVT Archives and Chemical Embedding Measures for Multi-Objective Quality Diversity in Molecular Design","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16586","citing_title":"A Systematic Survey and Benchmark of Deep Learning for Molecular Property Prediction in the Foundation Model Era","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ","json":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ.json","graph_json":"https://pith.science/api/pith-number/XFETXKGV3HJNNVROPCC7EQI2KJ/graph.json","events_json":"https://pith.science/api/pith-number/XFETXKGV3HJNNVROPCC7EQI2KJ/events.json","paper":"https://pith.science/paper/XFETXKGV"},"agent_actions":{"view_html":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ","download_json":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ.json","view_paper":"https://pith.science/paper/XFETXKGV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.01712&json=true","fetch_graph":"https://pith.science/api/pith-number/XFETXKGV3HJNNVROPCC7EQI2KJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XFETXKGV3HJNNVROPCC7EQI2KJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ/action/storage_attestation","attest_author":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ/action/author_attestation","sign_citation":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ/action/citation_signature","submit_replication":"https://pith.science/pith/XFETXKGV3HJNNVROPCC7EQI2KJ/action/replication_record"}},"created_at":"2026-07-05T04:54:33.034982+00:00","updated_at":"2026-07-05T04:54:33.034982+00:00"}