{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CDIR7H5Q3MFE2JFZD2ZAPZIMI2","short_pith_number":"pith:CDIR7H5Q","schema_version":"1.0","canonical_sha256":"10d11f9fb0db0a4d24b91eb207e50c46818a0dc9dfc0d71147695421fdf08e13","source":{"kind":"arxiv","id":"2506.02212","version":1},"attestation_state":"computed","paper":{"title":"Leveraging Natural Language Processing to Unravel the Mystery of Life: A Review of NLP Approaches in Genomics, Transcriptomics, and Proteomics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.GN"],"primary_cat":"cs.CL","authors_text":"David Burstein, Ella Rannon","submitted_at":"2025-06-02T19:54:03Z","abstract_excerpt":"Natural Language Processing (NLP) has transformed various fields beyond linguistics by applying techniques originally developed for human language to the analysis of biological sequences. This review explores the application of NLP methods to biological sequence data, focusing on genomics, transcriptomics, and proteomics. We examine how various NLP methods, from classic approaches like word2vec to advanced models employing transformers and hyena operators, are being adapted to analyze DNA, RNA, protein sequences, and entire genomes. The review also examines tokenization strategies and model ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02212","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:54:03Z","cross_cats_sorted":["cs.AI","q-bio.GN"],"title_canon_sha256":"95813acfb8db2db72f6a613b9b65a7a912a9bbbd56da84199b6ed1178fe1027c","abstract_canon_sha256":"68e73c873acb1dac654f3d4fe5b9b5e3f3e1214c8c641d733db77e287861ba87"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:34.484595Z","signature_b64":"dSbcxA3gWUh5iJeHAtXeS/iyG6dgFMabKeuYjTdgF3diXTru9e2Ok5NUKZkxaPOErePqQMS8QlFauPbROK43Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10d11f9fb0db0a4d24b91eb207e50c46818a0dc9dfc0d71147695421fdf08e13","last_reissued_at":"2026-07-05T11:14:34.484090Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:34.484090Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Natural Language Processing to Unravel the Mystery of Life: A Review of NLP Approaches in Genomics, Transcriptomics, and Proteomics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.GN"],"primary_cat":"cs.CL","authors_text":"David Burstein, Ella Rannon","submitted_at":"2025-06-02T19:54:03Z","abstract_excerpt":"Natural Language Processing (NLP) has transformed various fields beyond linguistics by applying techniques originally developed for human language to the analysis of biological sequences. This review explores the application of NLP methods to biological sequence data, focusing on genomics, transcriptomics, and proteomics. We examine how various NLP methods, from classic approaches like word2vec to advanced models employing transformers and hyena operators, are being adapted to analyze DNA, RNA, protein sequences, and entire genomes. The review also examines tokenization strategies and model ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02212","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02212","created_at":"2026-07-05T11:14:34.484149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02212v1","created_at":"2026-07-05T11:14:34.484149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02212","created_at":"2026-07-05T11:14:34.484149+00:00"},{"alias_kind":"pith_short_12","alias_value":"CDIR7H5Q3MFE","created_at":"2026-07-05T11:14:34.484149+00:00"},{"alias_kind":"pith_short_16","alias_value":"CDIR7H5Q3MFE2JFZ","created_at":"2026-07-05T11:14:34.484149+00:00"},{"alias_kind":"pith_short_8","alias_value":"CDIR7H5Q","created_at":"2026-07-05T11:14:34.484149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2","json":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2.json","graph_json":"https://pith.science/api/pith-number/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/graph.json","events_json":"https://pith.science/api/pith-number/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/events.json","paper":"https://pith.science/paper/CDIR7H5Q"},"agent_actions":{"view_html":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2","download_json":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2.json","view_paper":"https://pith.science/paper/CDIR7H5Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02212&json=true","fetch_graph":"https://pith.science/api/pith-number/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/graph.json","fetch_events":"https://pith.science/api/pith-number/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/action/storage_attestation","attest_author":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/action/author_attestation","sign_citation":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/action/citation_signature","submit_replication":"https://pith.science/pith/CDIR7H5Q3MFE2JFZD2ZAPZIMI2/action/replication_record"}},"created_at":"2026-07-05T11:14:34.484149+00:00","updated_at":"2026-07-05T11:14:34.484149+00:00"}