{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Z6N2EOUXGXUZUHIDFYNDT53RCA","short_pith_number":"pith:Z6N2EOUX","schema_version":"1.0","canonical_sha256":"cf9ba23a9735e99a1d032e1a39f7711002826f396a4bfd3253ed30f009167929","source":{"kind":"arxiv","id":"2112.03041","version":1},"attestation_state":"computed","paper":{"title":"Keeping it Simple: Language Models can learn Complex Molecular Distributions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Al\\'an Aspuru-Guzik, Daniel Flam-Shepherd, Kevin Zhu","submitted_at":"2021-12-06T13:40:58Z","abstract_excerpt":"Deep generative models of molecules have grown immensely in popularity, trained on relevant datasets, these models are used to search through chemical space. The downstream utility of generative models for the inverse design of novel functional compounds depends on their ability to learn a training distribution of molecules. The most simple example is a language model that takes the form of a recurrent neural network and generates molecules using a string representation. More sophisticated are graph generative models, which sequentially construct molecular graphs and typically achieve state of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.03041","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-06T13:40:58Z","cross_cats_sorted":["cs.AI","q-bio.QM"],"title_canon_sha256":"ed8eb7f1fee81e5cc33ad3f3ebbbd546e3e7caee9c975541b7fecc6788620d0d","abstract_canon_sha256":"6e6cd14bc1c587ae89e8ab281152956fe6667f8b910923dae4c9a1c4333fa2f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:08:37.714636Z","signature_b64":"S70jQYyLoF07TNOj3ei/TJbEK1Zxpu8ZwUNBxEb151R8ddsk73wmKVbc2BHL8SwC6Dm9p8PBq1eNZAm3jTnIAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf9ba23a9735e99a1d032e1a39f7711002826f396a4bfd3253ed30f009167929","last_reissued_at":"2026-07-05T06:08:37.714190Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:08:37.714190Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Keeping it Simple: Language Models can learn Complex Molecular Distributions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Al\\'an Aspuru-Guzik, Daniel Flam-Shepherd, Kevin Zhu","submitted_at":"2021-12-06T13:40:58Z","abstract_excerpt":"Deep generative models of molecules have grown immensely in popularity, trained on relevant datasets, these models are used to search through chemical space. The downstream utility of generative models for the inverse design of novel functional compounds depends on their ability to learn a training distribution of molecules. The most simple example is a language model that takes the form of a recurrent neural network and generates molecules using a string representation. More sophisticated are graph generative models, which sequentially construct molecular graphs and typically achieve state of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.03041","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.03041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.03041","created_at":"2026-07-05T06:08:37.714253+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.03041v1","created_at":"2026-07-05T06:08:37.714253+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.03041","created_at":"2026-07-05T06:08:37.714253+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z6N2EOUXGXUZ","created_at":"2026-07-05T06:08:37.714253+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z6N2EOUXGXUZUHID","created_at":"2026-07-05T06:08:37.714253+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z6N2EOUX","created_at":"2026-07-05T06:08:37.714253+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.13791","citing_title":"Physics Reasoner: Knowledge-Augmented Reasoning for Solving Physics Problems with Large Language Models","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA","json":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA.json","graph_json":"https://pith.science/api/pith-number/Z6N2EOUXGXUZUHIDFYNDT53RCA/graph.json","events_json":"https://pith.science/api/pith-number/Z6N2EOUXGXUZUHIDFYNDT53RCA/events.json","paper":"https://pith.science/paper/Z6N2EOUX"},"agent_actions":{"view_html":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA","download_json":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA.json","view_paper":"https://pith.science/paper/Z6N2EOUX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.03041&json=true","fetch_graph":"https://pith.science/api/pith-number/Z6N2EOUXGXUZUHIDFYNDT53RCA/graph.json","fetch_events":"https://pith.science/api/pith-number/Z6N2EOUXGXUZUHIDFYNDT53RCA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA/action/storage_attestation","attest_author":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA/action/author_attestation","sign_citation":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA/action/citation_signature","submit_replication":"https://pith.science/pith/Z6N2EOUXGXUZUHIDFYNDT53RCA/action/replication_record"}},"created_at":"2026-07-05T06:08:37.714253+00:00","updated_at":"2026-07-05T06:08:37.714253+00:00"}