{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:5IWOWTUQRVGLRGTR3LO2L37JPT","short_pith_number":"pith:5IWOWTUQ","schema_version":"1.0","canonical_sha256":"ea2ceb4e908d4cb89a71dadda5efe97ce0f5a54c591e14f50bf6e1ef1858b90b","source":{"kind":"arxiv","id":"2607.08404","version":1},"attestation_state":"computed","paper":{"title":"DrugGen 2: A disease-aware language model for enhancing drug discovery","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"q-bio.QM","authors_text":"Ali Motahharynia, Hajar Sirous, Mahsa Sheikholeslami, Matin Irajpour, Mohammadreza Ghaffarzadeh-Esfahani, Navid Mazrouei, Yousof Gheisari","submitted_at":"2026-07-09T12:29:33Z","abstract_excerpt":"Current computational approaches for drug design typically focus on generating molecules conditioned on specific targets or general molecular properties, often neglecting the influence of disease context on target behavior and therapeutic outcomes. To address this gap, we introduce DrugGen-2, a novel generative model that designs small molecules conditioned on both disease ontology and target protein sequences. DrugGen-2 was developed by fine-tuning a pre-trained GPT-2 model on a curated dataset of approved drugs linked to their diseases and targets, using a two-step strategy of supervised fin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.08404","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"q-bio.QM","submitted_at":"2026-07-09T12:29:33Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"c86a44bcd21451f45849b3563dc2bb1b4c4a6d44eca5000eaf29f9537ed997ac","abstract_canon_sha256":"0f5d44068f1048b3fbfb470aeaac2e09971f895bdcba85e027dc50152d7eaf8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-10T01:19:49.512085Z","signature_b64":"yYiZuqq50MJqY5Kk6aDtlUpzW8g8Ne9mIHe8ACUTBVDbIU+XFpoJgImo+VgaRHFZVDUvVBS7VyggCz30FQI5Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea2ceb4e908d4cb89a71dadda5efe97ce0f5a54c591e14f50bf6e1ef1858b90b","last_reissued_at":"2026-07-10T01:19:49.511620Z","signature_status":"signed_v1","first_computed_at":"2026-07-10T01:19:49.511620Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DrugGen 2: A disease-aware language model for enhancing drug discovery","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"q-bio.QM","authors_text":"Ali Motahharynia, Hajar Sirous, Mahsa Sheikholeslami, Matin Irajpour, Mohammadreza Ghaffarzadeh-Esfahani, Navid Mazrouei, Yousof Gheisari","submitted_at":"2026-07-09T12:29:33Z","abstract_excerpt":"Current computational approaches for drug design typically focus on generating molecules conditioned on specific targets or general molecular properties, often neglecting the influence of disease context on target behavior and therapeutic outcomes. To address this gap, we introduce DrugGen-2, a novel generative model that designs small molecules conditioned on both disease ontology and target protein sequences. DrugGen-2 was developed by fine-tuning a pre-trained GPT-2 model on a curated dataset of approved drugs linked to their diseases and targets, using a two-step strategy of supervised fin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.08404","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.08404/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.08404","created_at":"2026-07-10T01:19:49.511690+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.08404v1","created_at":"2026-07-10T01:19:49.511690+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.08404","created_at":"2026-07-10T01:19:49.511690+00:00"},{"alias_kind":"pith_short_12","alias_value":"5IWOWTUQRVGL","created_at":"2026-07-10T01:19:49.511690+00:00"},{"alias_kind":"pith_short_16","alias_value":"5IWOWTUQRVGLRGTR","created_at":"2026-07-10T01:19:49.511690+00:00"},{"alias_kind":"pith_short_8","alias_value":"5IWOWTUQ","created_at":"2026-07-10T01:19:49.511690+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT","json":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT.json","graph_json":"https://pith.science/api/pith-number/5IWOWTUQRVGLRGTR3LO2L37JPT/graph.json","events_json":"https://pith.science/api/pith-number/5IWOWTUQRVGLRGTR3LO2L37JPT/events.json","paper":"https://pith.science/paper/5IWOWTUQ"},"agent_actions":{"view_html":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT","download_json":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT.json","view_paper":"https://pith.science/paper/5IWOWTUQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.08404&json=true","fetch_graph":"https://pith.science/api/pith-number/5IWOWTUQRVGLRGTR3LO2L37JPT/graph.json","fetch_events":"https://pith.science/api/pith-number/5IWOWTUQRVGLRGTR3LO2L37JPT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT/action/storage_attestation","attest_author":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT/action/author_attestation","sign_citation":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT/action/citation_signature","submit_replication":"https://pith.science/pith/5IWOWTUQRVGLRGTR3LO2L37JPT/action/replication_record"}},"created_at":"2026-07-10T01:19:49.511690+00:00","updated_at":"2026-07-10T01:19:49.511690+00:00"}