{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JJ2IRMVZ2FMNVN754KFA2BDBHH","short_pith_number":"pith:JJ2IRMVZ","schema_version":"1.0","canonical_sha256":"4a7488b2b9d158dab7fde28a0d046139e70c882afcb3019ad6e164c4e0b3be67","source":{"kind":"arxiv","id":"2505.02206","version":1},"attestation_state":"computed","paper":{"title":"DNAZEN: Enhanced Gene Sequence Representations via Mixed Granularities of Coding Units","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Lei Mao, Yan Song, Yuanhe Tian","submitted_at":"2025-05-04T18:02:28Z","abstract_excerpt":"Genome modeling conventionally treats gene sequence as a language, reflecting its structured motifs and long-range dependencies analogous to linguistic units and organization principles such as words and syntax. Recent studies utilize advanced neural networks, ranging from convolutional and recurrent models to Transformer-based models, to capture contextual information of gene sequence, with the primary goal of obtaining effective gene sequence representations and thus enhance the models' understanding of various running gene samples. However, these approaches often directly apply language mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.02206","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-04T18:02:28Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"2de377d3368aaa205ffe5b6c3a3e88d2d034b959f6ac576aa097946edef6a6c0","abstract_canon_sha256":"bc32a5e68d6dc0d2cb9e73d68501510668e99cb0d9e6c091ae6c585a2d29ea59"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:33.669161Z","signature_b64":"0gVMZxIiaKunqzFr1TvTEJ5Z8mnTPu69ExLdkK2tlNVlHzUsrLauQOr/vXMBMCL0VxlA47fcRIve1+VE6OQjDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a7488b2b9d158dab7fde28a0d046139e70c882afcb3019ad6e164c4e0b3be67","last_reissued_at":"2026-07-05T10:58:33.668671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:33.668671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DNAZEN: Enhanced Gene Sequence Representations via Mixed Granularities of Coding Units","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Lei Mao, Yan Song, Yuanhe Tian","submitted_at":"2025-05-04T18:02:28Z","abstract_excerpt":"Genome modeling conventionally treats gene sequence as a language, reflecting its structured motifs and long-range dependencies analogous to linguistic units and organization principles such as words and syntax. Recent studies utilize advanced neural networks, ranging from convolutional and recurrent models to Transformer-based models, to capture contextual information of gene sequence, with the primary goal of obtaining effective gene sequence representations and thus enhance the models' understanding of various running gene samples. However, these approaches often directly apply language mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.02206","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.02206/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.02206","created_at":"2026-07-05T10:58:33.668735+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.02206v1","created_at":"2026-07-05T10:58:33.668735+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.02206","created_at":"2026-07-05T10:58:33.668735+00:00"},{"alias_kind":"pith_short_12","alias_value":"JJ2IRMVZ2FMN","created_at":"2026-07-05T10:58:33.668735+00:00"},{"alias_kind":"pith_short_16","alias_value":"JJ2IRMVZ2FMNVN75","created_at":"2026-07-05T10:58:33.668735+00:00"},{"alias_kind":"pith_short_8","alias_value":"JJ2IRMVZ","created_at":"2026-07-05T10:58:33.668735+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.15087","citing_title":"Evaluation of Coding Schemes for Transformer-based Gene Sequence Modeling","ref_index":2017,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH","json":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH.json","graph_json":"https://pith.science/api/pith-number/JJ2IRMVZ2FMNVN754KFA2BDBHH/graph.json","events_json":"https://pith.science/api/pith-number/JJ2IRMVZ2FMNVN754KFA2BDBHH/events.json","paper":"https://pith.science/paper/JJ2IRMVZ"},"agent_actions":{"view_html":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH","download_json":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH.json","view_paper":"https://pith.science/paper/JJ2IRMVZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.02206&json=true","fetch_graph":"https://pith.science/api/pith-number/JJ2IRMVZ2FMNVN754KFA2BDBHH/graph.json","fetch_events":"https://pith.science/api/pith-number/JJ2IRMVZ2FMNVN754KFA2BDBHH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH/action/storage_attestation","attest_author":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH/action/author_attestation","sign_citation":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH/action/citation_signature","submit_replication":"https://pith.science/pith/JJ2IRMVZ2FMNVN754KFA2BDBHH/action/replication_record"}},"created_at":"2026-07-05T10:58:33.668735+00:00","updated_at":"2026-07-05T10:58:33.668735+00:00"}