{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FNRHOYPNCA2GV3OZ6KX5NRDEFS","short_pith_number":"pith:FNRHOYPN","schema_version":"1.0","canonical_sha256":"2b627761ed10346aedd9f2afd6c4642cbfaad169226e69cde4f6a30739ff21f9","source":{"kind":"arxiv","id":"2306.15794","version":2},"attestation_state":"computed","paper":{"title":"HyenaDNA: Long-Range Genomic Sequence Modeling at Single Nucleotide Resolution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["q-bio.GN"],"primary_cat":"cs.LG","authors_text":"Aman Patel, Armin Thomas, Callum Birch-Sykes, Chris R\\'e, Clayton Rabideau, Eric Nguyen, Marjan Faizi, Michael Poli, Michael Wornow, Stefano Ermon, Stefano Massaroli, Stephen A. Baccus, Yoshua Bengio","submitted_at":"2023-06-27T20:46:34Z","abstract_excerpt":"Genomic (DNA) sequences encode an enormous amount of information for gene regulation and protein synthesis. Similar to natural language models, researchers have proposed foundation models in genomics to learn generalizable features from unlabeled genome data that can then be fine-tuned for downstream tasks such as identifying regulatory elements. Due to the quadratic scaling of attention, previous Transformer-based genomic models have used 512 to 4k tokens as context (<0.001% of the human genome), significantly limiting the modeling of long-range interactions in DNA. In addition, these methods"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.15794","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-27T20:46:34Z","cross_cats_sorted":["q-bio.GN"],"title_canon_sha256":"539636f5f0b9d2e032b9deb321e4b256f25622189e554d189b68c3401f86f615","abstract_canon_sha256":"dc47b411f621e38c3feba7a5ef2be9830bf07d48d133f7aa6c8e8485b4b5d4a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:17.371409Z","signature_b64":"e2eqImX89/8FG8t3KV0Gd0HTx1rw+Yn9L2vMDaXWx2WIbMp0c4UryQ1y623mmTT3NhV0k8rHlbSIAak5Sne5Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b627761ed10346aedd9f2afd6c4642cbfaad169226e69cde4f6a30739ff21f9","last_reissued_at":"2026-07-05T07:12:17.370906Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:17.370906Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HyenaDNA: Long-Range Genomic Sequence Modeling at Single Nucleotide Resolution","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["q-bio.GN"],"primary_cat":"cs.LG","authors_text":"Aman Patel, Armin Thomas, Callum Birch-Sykes, Chris R\\'e, Clayton Rabideau, Eric Nguyen, Marjan Faizi, Michael Poli, Michael Wornow, Stefano Ermon, Stefano Massaroli, Stephen A. Baccus, Yoshua Bengio","submitted_at":"2023-06-27T20:46:34Z","abstract_excerpt":"Genomic (DNA) sequences encode an enormous amount of information for gene regulation and protein synthesis. Similar to natural language models, researchers have proposed foundation models in genomics to learn generalizable features from unlabeled genome data that can then be fine-tuned for downstream tasks such as identifying regulatory elements. Due to the quadratic scaling of attention, previous Transformer-based genomic models have used 512 to 4k tokens as context (<0.001% of the human genome), significantly limiting the modeling of long-range interactions in DNA. In addition, these methods"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.15794","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.15794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.15794","created_at":"2026-07-05T07:12:17.370965+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.15794v2","created_at":"2026-07-05T07:12:17.370965+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.15794","created_at":"2026-07-05T07:12:17.370965+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNRHOYPNCA2G","created_at":"2026-07-05T07:12:17.370965+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNRHOYPNCA2GV3OZ","created_at":"2026-07-05T07:12:17.370965+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNRHOYPN","created_at":"2026-07-05T07:12:17.370965+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS","json":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS.json","graph_json":"https://pith.science/api/pith-number/FNRHOYPNCA2GV3OZ6KX5NRDEFS/graph.json","events_json":"https://pith.science/api/pith-number/FNRHOYPNCA2GV3OZ6KX5NRDEFS/events.json","paper":"https://pith.science/paper/FNRHOYPN"},"agent_actions":{"view_html":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS","download_json":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS.json","view_paper":"https://pith.science/paper/FNRHOYPN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.15794&json=true","fetch_graph":"https://pith.science/api/pith-number/FNRHOYPNCA2GV3OZ6KX5NRDEFS/graph.json","fetch_events":"https://pith.science/api/pith-number/FNRHOYPNCA2GV3OZ6KX5NRDEFS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS/action/storage_attestation","attest_author":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS/action/author_attestation","sign_citation":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS/action/citation_signature","submit_replication":"https://pith.science/pith/FNRHOYPNCA2GV3OZ6KX5NRDEFS/action/replication_record"}},"created_at":"2026-07-05T07:12:17.370965+00:00","updated_at":"2026-07-05T07:12:17.370965+00:00"}