{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2","short_pith_number":"pith:Q3HVZM4Z","schema_version":"1.0","canonical_sha256":"86cf5cb3999c29a389952a1165c3792690cf6d726a6d0efc4251f6bc7ef00ab4","source":{"kind":"arxiv","id":"2310.07276","version":3},"attestation_state":"computed","paper":{"title":"BioT5: Enriching Cross-modal Integration in Biology with Chemical Knowledge and Natural Language Associations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","q-bio.BM"],"primary_cat":"cs.CL","authors_text":"Jinhua Zhu, Kaiyuan Gao, Kehan Wu, Lijun Wu, Qizhi Pei, Rui Yan, Wei Zhang, Yingce Xia","submitted_at":"2023-10-11T07:57:08Z","abstract_excerpt":"Recent advancements in biological research leverage the integration of molecules, proteins, and natural language to enhance drug discovery. However, current models exhibit several limitations, such as the generation of invalid molecular SMILES, underutilization of contextual information, and equal treatment of structured and unstructured knowledge. To address these issues, we propose $\\mathbf{BioT5}$, a comprehensive pre-training framework that enriches cross-modal integration in biology with chemical knowledge and natural language associations. $\\mathbf{BioT5}$ utilizes SELFIES for $100%$ rob"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07276","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-11T07:57:08Z","cross_cats_sorted":["cs.AI","cs.LG","q-bio.BM"],"title_canon_sha256":"4ff0bee1259f1d875c9a24b54eec3f1abbeff2494f15d283888890b6a41c454a","abstract_canon_sha256":"8f18c641732e2f3d3036486451d9a3c1f9b1e3b71ae7a7e0e6710221e88f82a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:38:31.571241Z","signature_b64":"iBrlyjWzMBRPlHF07Qg4yhEPJhBI5B2rGdNPSO2rHrOzhX5/sCxvkl5bZrhGKHbmwxpMjvlM7aWapZpYhfLxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86cf5cb3999c29a389952a1165c3792690cf6d726a6d0efc4251f6bc7ef00ab4","last_reissued_at":"2026-07-05T07:38:31.570816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:38:31.570816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BioT5: Enriching Cross-modal Integration in Biology with Chemical Knowledge and Natural Language Associations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","q-bio.BM"],"primary_cat":"cs.CL","authors_text":"Jinhua Zhu, Kaiyuan Gao, Kehan Wu, Lijun Wu, Qizhi Pei, Rui Yan, Wei Zhang, Yingce Xia","submitted_at":"2023-10-11T07:57:08Z","abstract_excerpt":"Recent advancements in biological research leverage the integration of molecules, proteins, and natural language to enhance drug discovery. However, current models exhibit several limitations, such as the generation of invalid molecular SMILES, underutilization of contextual information, and equal treatment of structured and unstructured knowledge. To address these issues, we propose $\\mathbf{BioT5}$, a comprehensive pre-training framework that enriches cross-modal integration in biology with chemical knowledge and natural language associations. $\\mathbf{BioT5}$ utilizes SELFIES for $100%$ rob"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07276","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07276/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07276","created_at":"2026-07-05T07:38:31.570868+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07276v3","created_at":"2026-07-05T07:38:31.570868+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07276","created_at":"2026-07-05T07:38:31.570868+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q3HVZM4ZTQU2","created_at":"2026-07-05T07:38:31.570868+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q3HVZM4ZTQU2HCMV","created_at":"2026-07-05T07:38:31.570868+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q3HVZM4Z","created_at":"2026-07-05T07:38:31.570868+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.02231","citing_title":"SmileyLlama: Modifying Large Language Models for Directed Chemical Space Exploration","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22287","citing_title":"SciCore-Mol: Augmenting Large Language Models with Pluggable Molecular Cognition Modules","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2","json":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2.json","graph_json":"https://pith.science/api/pith-number/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/graph.json","events_json":"https://pith.science/api/pith-number/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/events.json","paper":"https://pith.science/paper/Q3HVZM4Z"},"agent_actions":{"view_html":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2","download_json":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2.json","view_paper":"https://pith.science/paper/Q3HVZM4Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07276&json=true","fetch_graph":"https://pith.science/api/pith-number/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/graph.json","fetch_events":"https://pith.science/api/pith-number/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/action/storage_attestation","attest_author":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/action/author_attestation","sign_citation":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/action/citation_signature","submit_replication":"https://pith.science/pith/Q3HVZM4ZTQU2HCMVFIIWLQ3ZE2/action/replication_record"}},"created_at":"2026-07-05T07:38:31.570868+00:00","updated_at":"2026-07-05T07:38:31.570868+00:00"}