{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YFZLKIDGCF5B47YVRRJIIMHIXC","short_pith_number":"pith:YFZLKIDG","schema_version":"1.0","canonical_sha256":"c172b52066117a1e7f158c528430e8b8b0b2d6407a24d5904b315646c58cb0b3","source":{"kind":"arxiv","id":"2311.12570","version":4},"attestation_state":"computed","paper":{"title":"BEND: Benchmarking DNA Language Models on biologically meaningful tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"q-bio.GN","authors_text":"Dennis Madsen, Dennis Pultz, Felix Teufel, Frederikke Isa Marin, Marc Horlacher, Ole Winther, Wouter Boomsma","submitted_at":"2023-11-21T12:34:00Z","abstract_excerpt":"The genome sequence contains the blueprint for governing cellular processes. While the availability of genomes has vastly increased over the last decades, experimental annotation of the various functional, non-coding and regulatory elements encoded in the DNA sequence remains both expensive and challenging. This has sparked interest in unsupervised language modeling of genomic DNA, a paradigm that has seen great success for protein sequence data. Although various DNA language models have been proposed, evaluation tasks often differ between individual works, and might not fully recapitulate the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.12570","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"q-bio.GN","submitted_at":"2023-11-21T12:34:00Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"58a18a464f7aa1618e7610c899523393c461bdcff4a93671dbdc3441e09f42d3","abstract_canon_sha256":"b1bc1311e1e5041ba3103266edfcfefe863c15eeceb3bf13ee774e115c6db28b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:52.239081Z","signature_b64":"nz3rBmEvC4ohlvH9ivOErYk71FVbqELHn1AuZKPWrDJz7KhOg55qVOpYW7RxT4MjVyqPBcPXwUVptS9ZZIV4BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c172b52066117a1e7f158c528430e8b8b0b2d6407a24d5904b315646c58cb0b3","last_reissued_at":"2026-07-05T08:05:52.238672Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:52.238672Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BEND: Benchmarking DNA Language Models on biologically meaningful tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"q-bio.GN","authors_text":"Dennis Madsen, Dennis Pultz, Felix Teufel, Frederikke Isa Marin, Marc Horlacher, Ole Winther, Wouter Boomsma","submitted_at":"2023-11-21T12:34:00Z","abstract_excerpt":"The genome sequence contains the blueprint for governing cellular processes. While the availability of genomes has vastly increased over the last decades, experimental annotation of the various functional, non-coding and regulatory elements encoded in the DNA sequence remains both expensive and challenging. This has sparked interest in unsupervised language modeling of genomic DNA, a paradigm that has seen great success for protein sequence data. Although various DNA language models have been proposed, evaluation tasks often differ between individual works, and might not fully recapitulate the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.12570","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.12570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.12570","created_at":"2026-07-05T08:05:52.238731+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.12570v4","created_at":"2026-07-05T08:05:52.238731+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.12570","created_at":"2026-07-05T08:05:52.238731+00:00"},{"alias_kind":"pith_short_12","alias_value":"YFZLKIDGCF5B","created_at":"2026-07-05T08:05:52.238731+00:00"},{"alias_kind":"pith_short_16","alias_value":"YFZLKIDGCF5B47YV","created_at":"2026-07-05T08:05:52.238731+00:00"},{"alias_kind":"pith_short_8","alias_value":"YFZLKIDG","created_at":"2026-07-05T08:05:52.238731+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.16570","citing_title":"In Search of Lost DNA Sequence Pretraining","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC","json":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC.json","graph_json":"https://pith.science/api/pith-number/YFZLKIDGCF5B47YVRRJIIMHIXC/graph.json","events_json":"https://pith.science/api/pith-number/YFZLKIDGCF5B47YVRRJIIMHIXC/events.json","paper":"https://pith.science/paper/YFZLKIDG"},"agent_actions":{"view_html":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC","download_json":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC.json","view_paper":"https://pith.science/paper/YFZLKIDG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.12570&json=true","fetch_graph":"https://pith.science/api/pith-number/YFZLKIDGCF5B47YVRRJIIMHIXC/graph.json","fetch_events":"https://pith.science/api/pith-number/YFZLKIDGCF5B47YVRRJIIMHIXC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC/action/storage_attestation","attest_author":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC/action/author_attestation","sign_citation":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC/action/citation_signature","submit_replication":"https://pith.science/pith/YFZLKIDGCF5B47YVRRJIIMHIXC/action/replication_record"}},"created_at":"2026-07-05T08:05:52.238731+00:00","updated_at":"2026-07-05T08:05:52.238731+00:00"}