{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:3DVZQ4KAJ6PKGNWMNF5HC2PWWF","short_pith_number":"pith:3DVZQ4KA","schema_version":"1.0","canonical_sha256":"d8eb9871404f9ea336cc697a7169f6b16dfb524b022a2984eaa0128b2bafa684","source":{"kind":"arxiv","id":"2010.09885","version":2},"attestation_state":"computed","paper":{"title":"ChemBERTa: Large-Scale Self-Supervised Pretraining for Molecular Property Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","physics.chem-ph","q-bio.BM"],"primary_cat":"cs.LG","authors_text":"Bharath Ramsundar, Gabriel Grand, Seyone Chithrananda","submitted_at":"2020-10-19T21:41:41Z","abstract_excerpt":"GNNs and chemical fingerprints are the predominant approaches to representing molecules for property prediction. However, in NLP, transformers have become the de-facto standard for representation learning thanks to their strong downstream task transfer. In parallel, the software ecosystem around transformers is maturing rapidly, with libraries like HuggingFace and BertViz enabling streamlined training and introspection. In this work, we make one of the first attempts to systematically evaluate transformers on molecular property prediction tasks via our ChemBERTa model. ChemBERTa scales well wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.09885","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-19T21:41:41Z","cross_cats_sorted":["cs.CL","physics.chem-ph","q-bio.BM"],"title_canon_sha256":"24f6b91f09cf8e563295dffb925fbd89bed1189759091f859fbe7ab7582bb175","abstract_canon_sha256":"5e395ad3e3bafe2cfbf18b1b16ac1cc5da2213162ed19c41d483a2c3db4db240"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:31.355812Z","signature_b64":"IUh3UKpOFkDYF040d4UmjY0qisbscjGSONQK4K0w3Q2xulmo0L+L9+oQKjgf1Lv7ckJQC7NDXSJSzNLDsOGNDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d8eb9871404f9ea336cc697a7169f6b16dfb524b022a2984eaa0128b2bafa684","last_reissued_at":"2026-07-05T01:45:31.355328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:31.355328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChemBERTa: Large-Scale Self-Supervised Pretraining for Molecular Property Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","physics.chem-ph","q-bio.BM"],"primary_cat":"cs.LG","authors_text":"Bharath Ramsundar, Gabriel Grand, Seyone Chithrananda","submitted_at":"2020-10-19T21:41:41Z","abstract_excerpt":"GNNs and chemical fingerprints are the predominant approaches to representing molecules for property prediction. However, in NLP, transformers have become the de-facto standard for representation learning thanks to their strong downstream task transfer. In parallel, the software ecosystem around transformers is maturing rapidly, with libraries like HuggingFace and BertViz enabling streamlined training and introspection. In this work, we make one of the first attempts to systematically evaluate transformers on molecular property prediction tasks via our ChemBERTa model. ChemBERTa scales well wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.09885","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.09885/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.09885","created_at":"2026-07-05T01:45:31.355388+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.09885v2","created_at":"2026-07-05T01:45:31.355388+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.09885","created_at":"2026-07-05T01:45:31.355388+00:00"},{"alias_kind":"pith_short_12","alias_value":"3DVZQ4KAJ6PK","created_at":"2026-07-05T01:45:31.355388+00:00"},{"alias_kind":"pith_short_16","alias_value":"3DVZQ4KAJ6PKGNWM","created_at":"2026-07-05T01:45:31.355388+00:00"},{"alias_kind":"pith_short_8","alias_value":"3DVZQ4KA","created_at":"2026-07-05T01:45:31.355388+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":34,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05736","citing_title":"Multimodal Molecular Representation Learning with Graph Neural Networks, Deep & Cross Networks, and SMILES Embeddings","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06605","citing_title":"A Quiet Failure in Calibrated Virtual Screening: Marginal Conformal Prediction Under-Covers the Minority Class, and a Class-Conditional Fix Recovers It","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22731","citing_title":"Closed-loop Auto Research for Molecular Property Prediction: Discovering and Certifying Generalizable Improvements","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23443","citing_title":"What Does a Chemical Language Model Know About Molecules?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20756","citing_title":"A large-scale foundation model enables simulation-to-real adaptation for nuclear magnetic resonance-based molecular structure analysis","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18703","citing_title":"Contextualizing Biological Language Models across Modalities via Logit-Space Contrastive Alignment","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02140","citing_title":"Probing Chemical Language Models: Effects of Pre-training and Fine-tuning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12113","citing_title":"Augmenting Molecular Language Models with Local $n$-gram Memory","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05693","citing_title":"MolE-RAG: Molecular Structure-Enhanced Retrieval-Augmented Generation for Chemistry","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02106","citing_title":"When Tabular Foundation Models Transfer Across Modalities: A Systematic Evaluation Across 95 Datasets, 7 Modalities, and Two Regimes","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30695","citing_title":"Modeling Cell-Cycle-Aware Single-Cell Drug Perturbation Responses","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29776","citing_title":"Towards Generalizable and Evidential Nuclear Magnetic Resonance-Based Molecular Structure Elucidation via Large Language Model Agent","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28655","citing_title":"AutoScientists: Self-Organizing Agent Teams for Long-Running Scientific Experimentation","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11382","citing_title":"GLACIER: A Multimodal Student-Teacher Foundation Model for Molecular Property Prediction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18703","citing_title":"Contextualizing Biological Language Models across Modalities via Logit-Space Contrastive Alignment","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2407.09514","citing_title":"Machine Learning Based Prediction of Proton Conductivity in Metal-Organic Frameworks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2409.06080","citing_title":"Regression with Large Language Models for Materials and Molecular Property Prediction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20885","citing_title":"Training distribution determines the ceiling of drug-blind cancer sensitivity prediction","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26498","citing_title":"Do Larger Models Really Win in Drug Discovery? A Benchmark Assessment of Model Scaling in AI-Driven Molecular Property and Activity Prediction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2506.00239","citing_title":"SmellNet: A Large-scale Dataset for Real-world Smell Recognition","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18900","citing_title":"Foundation Models for Discovery and Exploration in Chemical Space","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22822","citing_title":"FlexMS is a flexible framework for benchmarking deep learning-based mass spectrum prediction tools in metabolomics","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2304.05376","citing_title":"ChemCrow: Augmenting large-language models with chemistry tools","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13262","citing_title":"Chem-GMNet: A Sphere-Native Geometric Transformer for Molecular Property Prediction","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26498","citing_title":"Do Larger Models Really Win in Drug Discovery? A Benchmark Assessment of Model Scaling in AI-Driven Molecular Property and Activity Prediction","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF","json":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF.json","graph_json":"https://pith.science/api/pith-number/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/graph.json","events_json":"https://pith.science/api/pith-number/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/events.json","paper":"https://pith.science/paper/3DVZQ4KA"},"agent_actions":{"view_html":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF","download_json":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF.json","view_paper":"https://pith.science/paper/3DVZQ4KA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.09885&json=true","fetch_graph":"https://pith.science/api/pith-number/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/graph.json","fetch_events":"https://pith.science/api/pith-number/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/action/storage_attestation","attest_author":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/action/author_attestation","sign_citation":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/action/citation_signature","submit_replication":"https://pith.science/pith/3DVZQ4KAJ6PKGNWMNF5HC2PWWF/action/replication_record"}},"created_at":"2026-07-05T01:45:31.355388+00:00","updated_at":"2026-07-05T01:45:31.355388+00:00"}