{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:OTHKJOYCHMXUCMSJ6PVRBNNXUB","short_pith_number":"pith:OTHKJOYC","schema_version":"1.0","canonical_sha256":"74cea4bb023b2f413249f3eb10b5b7a04b493dc88ea7c33ac9fd901346d6e11d","source":{"kind":"arxiv","id":"2607.19262","version":1},"attestation_state":"computed","paper":{"title":"BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Amanda Darling, Arjun Banerjee, Bryan Tegomoh, Claire Duvallet, David Stern, Dianzhuo Wang, Evan Seeyave, Harmon Bhasin, Joshua Stallings, Kenny Workman, Kevin Flyangolts, Shawn Higdon","submitted_at":"2026-07-21T16:33:57Z","abstract_excerpt":"As pathogen genomic surveillance scales, the bottleneck is shifting from data generation to analysis. We present BioSecBench-Surveillance, a verifiable benchmark of 100 evaluations testing whether AI agents can infer the right analysis pipeline from raw sequencing data and surveillance context. Each evaluation gives an agent only the data and context a human analyst would have, then grades its structured answer deterministically. The tasks span seven categories, from taxonomic classification to genetic-engineering detection, across diverse sample types and sequencing technologies. Across 3,962"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.19262","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-21T16:33:57Z","cross_cats_sorted":[],"title_canon_sha256":"ff8cd4df3896469c4ad2cea3c92e9d743968ce71fd2e5ec02cb555584104db34","abstract_canon_sha256":"84ef55d89417d95a15f66b1b83bd2a6130645cb6a6eb1dabdc0a584f300b4262"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-22T01:24:17.141628Z","signature_b64":"yYXzNKIGFb+pZJPz39/z62lNJx1n8+ePqug0P8obGj9hC1YLm+gsraXWiueWS68etNQ+LNKnCvVvIOIw5M3yDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74cea4bb023b2f413249f3eb10b5b7a04b493dc88ea7c33ac9fd901346d6e11d","last_reissued_at":"2026-07-22T01:24:17.140795Z","signature_status":"signed_v1","first_computed_at":"2026-07-22T01:24:17.140795Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Amanda Darling, Arjun Banerjee, Bryan Tegomoh, Claire Duvallet, David Stern, Dianzhuo Wang, Evan Seeyave, Harmon Bhasin, Joshua Stallings, Kenny Workman, Kevin Flyangolts, Shawn Higdon","submitted_at":"2026-07-21T16:33:57Z","abstract_excerpt":"As pathogen genomic surveillance scales, the bottleneck is shifting from data generation to analysis. We present BioSecBench-Surveillance, a verifiable benchmark of 100 evaluations testing whether AI agents can infer the right analysis pipeline from raw sequencing data and surveillance context. Each evaluation gives an agent only the data and context a human analyst would have, then grades its structured answer deterministically. The tasks span seven categories, from taxonomic classification to genetic-engineering detection, across diverse sample types and sequencing technologies. Across 3,962"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.19262","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.19262/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.19262","created_at":"2026-07-22T01:24:17.141217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.19262v1","created_at":"2026-07-22T01:24:17.141217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.19262","created_at":"2026-07-22T01:24:17.141217+00:00"},{"alias_kind":"pith_short_12","alias_value":"OTHKJOYCHMXU","created_at":"2026-07-22T01:24:17.141217+00:00"},{"alias_kind":"pith_short_16","alias_value":"OTHKJOYCHMXUCMSJ","created_at":"2026-07-22T01:24:17.141217+00:00"},{"alias_kind":"pith_short_8","alias_value":"OTHKJOYC","created_at":"2026-07-22T01:24:17.141217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB","json":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB.json","graph_json":"https://pith.science/api/pith-number/OTHKJOYCHMXUCMSJ6PVRBNNXUB/graph.json","events_json":"https://pith.science/api/pith-number/OTHKJOYCHMXUCMSJ6PVRBNNXUB/events.json","paper":"https://pith.science/paper/OTHKJOYC"},"agent_actions":{"view_html":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB","download_json":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB.json","view_paper":"https://pith.science/paper/OTHKJOYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.19262&json=true","fetch_graph":"https://pith.science/api/pith-number/OTHKJOYCHMXUCMSJ6PVRBNNXUB/graph.json","fetch_events":"https://pith.science/api/pith-number/OTHKJOYCHMXUCMSJ6PVRBNNXUB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB/action/storage_attestation","attest_author":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB/action/author_attestation","sign_citation":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB/action/citation_signature","submit_replication":"https://pith.science/pith/OTHKJOYCHMXUCMSJ6PVRBNNXUB/action/replication_record"}},"created_at":"2026-07-22T01:24:17.141217+00:00","updated_at":"2026-07-22T01:24:17.141217+00:00"}