{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VT6HXXH5572R3SOCMCZBW2CW2W","short_pith_number":"pith:VT6HXXH5","schema_version":"1.0","canonical_sha256":"acfc7bdcfdeff51dc9c260b21b6856d5a65b877f99bb75b7e64c330ec4616176","source":{"kind":"arxiv","id":"2401.04155","version":2},"attestation_state":"computed","paper":{"title":"Advancing bioinformatics with large language models: components, applications and perspectives","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"q-bio.QM","authors_text":"Haixia Xu, Jiajia Liu, Kang Li, Mengyuan Yang, Tiangang Wang, Xiaobo Zhou, Yankai Yu","submitted_at":"2024-01-08T17:26:59Z","abstract_excerpt":"Large language models (LLMs) are a class of artificial intelligence models based on deep learning, which have great performance in various tasks, especially in natural language processing (NLP). Large language models typically consist of artificial neural networks with numerous parameters, trained on large amounts of unlabeled input using self-supervised or semi-supervised learning. However, their potential for solving bioinformatics problems may even exceed their proficiency in modeling human language. In this review, we will provide a comprehensive overview of the essential components of lar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.04155","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"q-bio.QM","submitted_at":"2024-01-08T17:26:59Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5832e83d7e62b92507b1c2a8a9b2f2084c1f02f16565ed3f50048457d8bbff17","abstract_canon_sha256":"a81f46cd8c21c4e405355c85e28ba6c26cc397a5e1f18bc724c3f5a481f10a3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:13.586739Z","signature_b64":"ehGykhqCUUNwmzSxNpPKqyBX1dKF91EiymBoglpmCD/vEQ+OA/kdVhhjvdN8Pgk1jgWFuco9NVrAuBCAL+AbAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"acfc7bdcfdeff51dc9c260b21b6856d5a65b877f99bb75b7e64c330ec4616176","last_reissued_at":"2026-07-05T10:08:13.586198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:13.586198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Advancing bioinformatics with large language models: components, applications and perspectives","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"q-bio.QM","authors_text":"Haixia Xu, Jiajia Liu, Kang Li, Mengyuan Yang, Tiangang Wang, Xiaobo Zhou, Yankai Yu","submitted_at":"2024-01-08T17:26:59Z","abstract_excerpt":"Large language models (LLMs) are a class of artificial intelligence models based on deep learning, which have great performance in various tasks, especially in natural language processing (NLP). Large language models typically consist of artificial neural networks with numerous parameters, trained on large amounts of unlabeled input using self-supervised or semi-supervised learning. However, their potential for solving bioinformatics problems may even exceed their proficiency in modeling human language. In this review, we will provide a comprehensive overview of the essential components of lar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.04155","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.04155/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.04155","created_at":"2026-07-05T10:08:13.586258+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.04155v2","created_at":"2026-07-05T10:08:13.586258+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.04155","created_at":"2026-07-05T10:08:13.586258+00:00"},{"alias_kind":"pith_short_12","alias_value":"VT6HXXH5572R","created_at":"2026-07-05T10:08:13.586258+00:00"},{"alias_kind":"pith_short_16","alias_value":"VT6HXXH5572R3SOC","created_at":"2026-07-05T10:08:13.586258+00:00"},{"alias_kind":"pith_short_8","alias_value":"VT6HXXH5","created_at":"2026-07-05T10:08:13.586258+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.05944","citing_title":"Multi-granular Training Strategies for Robust Multi-hop Reasoning Over Noisy and Heterogeneous Knowledge Sources","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W","json":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W.json","graph_json":"https://pith.science/api/pith-number/VT6HXXH5572R3SOCMCZBW2CW2W/graph.json","events_json":"https://pith.science/api/pith-number/VT6HXXH5572R3SOCMCZBW2CW2W/events.json","paper":"https://pith.science/paper/VT6HXXH5"},"agent_actions":{"view_html":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W","download_json":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W.json","view_paper":"https://pith.science/paper/VT6HXXH5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.04155&json=true","fetch_graph":"https://pith.science/api/pith-number/VT6HXXH5572R3SOCMCZBW2CW2W/graph.json","fetch_events":"https://pith.science/api/pith-number/VT6HXXH5572R3SOCMCZBW2CW2W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W/action/storage_attestation","attest_author":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W/action/author_attestation","sign_citation":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W/action/citation_signature","submit_replication":"https://pith.science/pith/VT6HXXH5572R3SOCMCZBW2CW2W/action/replication_record"}},"created_at":"2026-07-05T10:08:13.586258+00:00","updated_at":"2026-07-05T10:08:13.586258+00:00"}