{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CZRXPYKJBD3AKHERJ5BXMXHMMJ","short_pith_number":"pith:CZRXPYKJ","schema_version":"1.0","canonical_sha256":"166377e14908f6051c914f43765cec626c3326be79ba3cab271b60d4736d3a5c","source":{"kind":"arxiv","id":"2505.06108","version":3},"attestation_state":"computed","paper":{"title":"LLMs Outperform Experts on Challenging Biology Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Lennart Justen","submitted_at":"2025-05-09T15:05:57Z","abstract_excerpt":"This study systematically evaluates 27 frontier Large Language Models on eight biology benchmarks spanning molecular biology, genetics, cloning, virology, and biosecurity. Models from major AI developers released between November 2022 and April 2025 were assessed through ten independent runs per benchmark. The findings reveal dramatic improvements in biological capabilities. Top model performance increased more than 4-fold on the challenging text-only subset of the Virology Capabilities Test over the study period, with OpenAI's o3 now performing twice as well as expert virologists. Several mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.06108","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-09T15:05:57Z","cross_cats_sorted":["cs.AI","q-bio.QM"],"title_canon_sha256":"3deb58a57d6911f272ce2dba86d99919d84ed3888848573b1cbd394a14e4a28f","abstract_canon_sha256":"d24ba62724429934ab73df3c3f169f3eec81f1cf4bb607ee1fbc67a9fafd1a73"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:24.416245Z","signature_b64":"ofo+03ZAEaErMo2ipc7YrdkyU7GrQwX8nO4HiqBwg5dyBRDZ+l5c3+MXuNRnmi6m8U8R9k2ZjwnBGdBkLMREDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"166377e14908f6051c914f43765cec626c3326be79ba3cab271b60d4736d3a5c","last_reissued_at":"2026-07-05T11:07:24.415775Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:24.415775Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs Outperform Experts on Challenging Biology Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Lennart Justen","submitted_at":"2025-05-09T15:05:57Z","abstract_excerpt":"This study systematically evaluates 27 frontier Large Language Models on eight biology benchmarks spanning molecular biology, genetics, cloning, virology, and biosecurity. Models from major AI developers released between November 2022 and April 2025 were assessed through ten independent runs per benchmark. The findings reveal dramatic improvements in biological capabilities. Top model performance increased more than 4-fold on the challenging text-only subset of the Virology Capabilities Test over the study period, with OpenAI's o3 now performing twice as well as expert virologists. Several mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.06108","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.06108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.06108","created_at":"2026-07-05T11:07:24.415834+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.06108v3","created_at":"2026-07-05T11:07:24.415834+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.06108","created_at":"2026-07-05T11:07:24.415834+00:00"},{"alias_kind":"pith_short_12","alias_value":"CZRXPYKJBD3A","created_at":"2026-07-05T11:07:24.415834+00:00"},{"alias_kind":"pith_short_16","alias_value":"CZRXPYKJBD3AKHER","created_at":"2026-07-05T11:07:24.415834+00:00"},{"alias_kind":"pith_short_8","alias_value":"CZRXPYKJ","created_at":"2026-07-05T11:07:24.415834+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01145","citing_title":"Reasoning4Sciences: Bridging Reasoning Language Models to All Scientific Branches","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01145","citing_title":"Reasoning4Sciences: Bridging Reasoning Language Models to All Scientific Branches","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17324","citing_title":"ASPI: Seeking Ambiguity Clarification Amplifies Prompt Injection Vulnerability in LLM Agents","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00267","citing_title":"Jailbroken Frontier Models Retain Their Capabilities","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10718","citing_title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ","json":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ.json","graph_json":"https://pith.science/api/pith-number/CZRXPYKJBD3AKHERJ5BXMXHMMJ/graph.json","events_json":"https://pith.science/api/pith-number/CZRXPYKJBD3AKHERJ5BXMXHMMJ/events.json","paper":"https://pith.science/paper/CZRXPYKJ"},"agent_actions":{"view_html":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ","download_json":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ.json","view_paper":"https://pith.science/paper/CZRXPYKJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.06108&json=true","fetch_graph":"https://pith.science/api/pith-number/CZRXPYKJBD3AKHERJ5BXMXHMMJ/graph.json","fetch_events":"https://pith.science/api/pith-number/CZRXPYKJBD3AKHERJ5BXMXHMMJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ/action/storage_attestation","attest_author":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ/action/author_attestation","sign_citation":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ/action/citation_signature","submit_replication":"https://pith.science/pith/CZRXPYKJBD3AKHERJ5BXMXHMMJ/action/replication_record"}},"created_at":"2026-07-05T11:07:24.415834+00:00","updated_at":"2026-07-05T11:07:24.415834+00:00"}