{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CFZIORHRW6WHQKRNI3K7VODSQV","short_pith_number":"pith:CFZIORHR","schema_version":"1.0","canonical_sha256":"11728744f1b7ac782a2d46d5fab87285404e81cb67c48fe7a6aa599db0ab1384","source":{"kind":"arxiv","id":"2402.08777","version":3},"attestation_state":"computed","paper":{"title":"DNABERT-S: Pioneering Species Differentiation with Species-Aware DNA Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CE","cs.CL"],"primary_cat":"q-bio.GN","authors_text":"Han Liu, Harrison Ho, Jiayi Wang, Lizhen Shi, Ramana V Davuluri, Weimin Wu, Zhihan Zhou, Zhong Wang","submitted_at":"2024-02-13T20:21:29Z","abstract_excerpt":"We introduce DNABERT-S, a tailored genome model that develops species-aware embeddings to naturally cluster and segregate DNA sequences of different species in the embedding space. Differentiating species from genomic sequences (i.e., DNA and RNA) is vital yet challenging, since many real-world species remain uncharacterized, lacking known genomes for reference. Embedding-based methods are therefore used to differentiate species in an unsupervised manner. DNABERT-S builds upon a pre-trained genome foundation model named DNABERT-2. To encourage effective embeddings to error-prone long-read DNA "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08777","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"q-bio.GN","submitted_at":"2024-02-13T20:21:29Z","cross_cats_sorted":["cs.AI","cs.CE","cs.CL"],"title_canon_sha256":"d7b94d6cfe80fbdda0b6ab4bb11863fa4c76a3dbc815ec0fc93ee9d5be19be8f","abstract_canon_sha256":"4a41d7a5bf5d5e0bdbeb838b19f723ec73e29035452d7e3b7adf1f93a6e77b4f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:47.261540Z","signature_b64":"RkEcZvKodXR55kOYpC3K/+zQqkJRyDw+AlhfyyptaNl58TnhR5xxxbHmq7cFgslgJ7yJ+ldYjZbxywRl9Y9jBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11728744f1b7ac782a2d46d5fab87285404e81cb67c48fe7a6aa599db0ab1384","last_reissued_at":"2026-07-05T09:23:47.260947Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:47.260947Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DNABERT-S: Pioneering Species Differentiation with Species-Aware DNA Embeddings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CE","cs.CL"],"primary_cat":"q-bio.GN","authors_text":"Han Liu, Harrison Ho, Jiayi Wang, Lizhen Shi, Ramana V Davuluri, Weimin Wu, Zhihan Zhou, Zhong Wang","submitted_at":"2024-02-13T20:21:29Z","abstract_excerpt":"We introduce DNABERT-S, a tailored genome model that develops species-aware embeddings to naturally cluster and segregate DNA sequences of different species in the embedding space. Differentiating species from genomic sequences (i.e., DNA and RNA) is vital yet challenging, since many real-world species remain uncharacterized, lacking known genomes for reference. Embedding-based methods are therefore used to differentiate species in an unsupervised manner. DNABERT-S builds upon a pre-trained genome foundation model named DNABERT-2. To encourage effective embeddings to error-prone long-read DNA "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08777","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08777/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08777","created_at":"2026-07-05T09:23:47.261040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08777v3","created_at":"2026-07-05T09:23:47.261040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08777","created_at":"2026-07-05T09:23:47.261040+00:00"},{"alias_kind":"pith_short_12","alias_value":"CFZIORHRW6WH","created_at":"2026-07-05T09:23:47.261040+00:00"},{"alias_kind":"pith_short_16","alias_value":"CFZIORHRW6WHQKRN","created_at":"2026-07-05T09:23:47.261040+00:00"},{"alias_kind":"pith_short_8","alias_value":"CFZIORHR","created_at":"2026-07-05T09:23:47.261040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12286","citing_title":"Set-Aggregated Genome Embeddings for Microbiome Abundance Prediction","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV","json":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV.json","graph_json":"https://pith.science/api/pith-number/CFZIORHRW6WHQKRNI3K7VODSQV/graph.json","events_json":"https://pith.science/api/pith-number/CFZIORHRW6WHQKRNI3K7VODSQV/events.json","paper":"https://pith.science/paper/CFZIORHR"},"agent_actions":{"view_html":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV","download_json":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV.json","view_paper":"https://pith.science/paper/CFZIORHR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08777&json=true","fetch_graph":"https://pith.science/api/pith-number/CFZIORHRW6WHQKRNI3K7VODSQV/graph.json","fetch_events":"https://pith.science/api/pith-number/CFZIORHRW6WHQKRNI3K7VODSQV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV/action/storage_attestation","attest_author":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV/action/author_attestation","sign_citation":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV/action/citation_signature","submit_replication":"https://pith.science/pith/CFZIORHRW6WHQKRNI3K7VODSQV/action/replication_record"}},"created_at":"2026-07-05T09:23:47.261040+00:00","updated_at":"2026-07-05T09:23:47.261040+00:00"}