{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NQAMGLEUAZI5ERC7LMSNC32N34","short_pith_number":"pith:NQAMGLEU","schema_version":"1.0","canonical_sha256":"6c00c32c940651d2445f5b24d16f4ddf18a34476b7b2f14189c0ed249c7b6203","source":{"kind":"arxiv","id":"2406.15534","version":1},"attestation_state":"computed","paper":{"title":"Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Hongyu Zhao, Hua Xu, Tianyu Liu, W. Jim Zheng, Xiao Luo, Yijia Xiao","submitted_at":"2024-06-21T14:19:10Z","abstract_excerpt":"The applications of large language models (LLMs) are promising for biomedical and healthcare research. Despite the availability of open-source LLMs trained using a wide range of biomedical data, current research on the applications of LLMs to genomics and proteomics is still limited. To fill this gap, we propose a collection of finetuned LLMs and multimodal LLMs (MLLMs), known as Geneverse, for three novel tasks in genomic and proteomic research. The models in Geneverse are trained and evaluated based on domain-specific datasets, and we use advanced parameter-efficient finetuning techniques to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15534","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-21T14:19:10Z","cross_cats_sorted":["cs.AI","cs.CL","q-bio.QM"],"title_canon_sha256":"4c5c3414dbb32e079e9f1fad4b71fd35d37504c2f5e2204ea806637491b768aa","abstract_canon_sha256":"692e1a9cdf52938faa9578dad83d75e407a5c5e46848655b3ae88880fcea329f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:48.527637Z","signature_b64":"CJgZltabZwbWChI83NY+RKWODKc3DQQhtrQ8+7Ay7kCynKncbFGOJN4FuKwDpWSDxgmdEmvpY/kjt3cSz4bkBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c00c32c940651d2445f5b24d16f4ddf18a34476b7b2f14189c0ed249c7b6203","last_reissued_at":"2026-07-05T09:10:48.527212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:48.527212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Hongyu Zhao, Hua Xu, Tianyu Liu, W. Jim Zheng, Xiao Luo, Yijia Xiao","submitted_at":"2024-06-21T14:19:10Z","abstract_excerpt":"The applications of large language models (LLMs) are promising for biomedical and healthcare research. Despite the availability of open-source LLMs trained using a wide range of biomedical data, current research on the applications of LLMs to genomics and proteomics is still limited. To fill this gap, we propose a collection of finetuned LLMs and multimodal LLMs (MLLMs), known as Geneverse, for three novel tasks in genomic and proteomic research. The models in Geneverse are trained and evaluated based on domain-specific datasets, and we use advanced parameter-efficient finetuning techniques to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15534","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15534/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15534","created_at":"2026-07-05T09:10:48.527264+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15534v1","created_at":"2026-07-05T09:10:48.527264+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15534","created_at":"2026-07-05T09:10:48.527264+00:00"},{"alias_kind":"pith_short_12","alias_value":"NQAMGLEUAZI5","created_at":"2026-07-05T09:10:48.527264+00:00"},{"alias_kind":"pith_short_16","alias_value":"NQAMGLEUAZI5ERC7","created_at":"2026-07-05T09:10:48.527264+00:00"},{"alias_kind":"pith_short_8","alias_value":"NQAMGLEU","created_at":"2026-07-05T09:10:48.527264+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.03360","citing_title":"A Multimodal, Multilingual, and Multidimensional Pipeline for Fine-grained Crowdsourcing Earthquake Damage Evaluation","ref_index":2023,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34","json":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34.json","graph_json":"https://pith.science/api/pith-number/NQAMGLEUAZI5ERC7LMSNC32N34/graph.json","events_json":"https://pith.science/api/pith-number/NQAMGLEUAZI5ERC7LMSNC32N34/events.json","paper":"https://pith.science/paper/NQAMGLEU"},"agent_actions":{"view_html":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34","download_json":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34.json","view_paper":"https://pith.science/paper/NQAMGLEU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15534&json=true","fetch_graph":"https://pith.science/api/pith-number/NQAMGLEUAZI5ERC7LMSNC32N34/graph.json","fetch_events":"https://pith.science/api/pith-number/NQAMGLEUAZI5ERC7LMSNC32N34/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34/action/storage_attestation","attest_author":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34/action/author_attestation","sign_citation":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34/action/citation_signature","submit_replication":"https://pith.science/pith/NQAMGLEUAZI5ERC7LMSNC32N34/action/replication_record"}},"created_at":"2026-07-05T09:10:48.527264+00:00","updated_at":"2026-07-05T09:10:48.527264+00:00"}