{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7VRO7PEPRJTIU2Q3JFH5JNT2TG","short_pith_number":"pith:7VRO7PEP","schema_version":"1.0","canonical_sha256":"fd62efbc8f8a668a6a1b494fd4b67a99972c8818cbe6c53d14ccd97ccdecb7a2","source":{"kind":"arxiv","id":"2505.19915","version":2},"attestation_state":"computed","paper":{"title":"Evaluating AI cyber capabilities with crowdsourced elicitation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Artem Petrov, Dmitrii Volkov","submitted_at":"2025-05-26T12:40:32Z","abstract_excerpt":"As AI systems become increasingly capable, understanding their offensive cyber potential is critical for informed governance and responsible deployment. However, it's hard to accurately bound their capabilities, and some prior evaluations dramatically underestimated them. The art of extracting maximum task-specific performance from AIs is called \"AI elicitation\", and today's safety organizations typically conduct it in-house. In this paper, we explore crowdsourcing elicitation efforts as an alternative to in-house elicitation work.\n  We host open-access AI tracks at two Capture The Flag (CTF) "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.19915","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-05-26T12:40:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9b8f3ca862ef2faafb52b1cd5f70c4bc9688200d9c6bd53bada5c9d5fe1d8594","abstract_canon_sha256":"c0eb94cebe02264c32ed9605f08093a99e87af128edfd1c0c8eb3bd4f34b5dd0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:38.393054Z","signature_b64":"S3hLHiJoxdRZ0GvPv6QvUGuTMS3VLF+/D31l4Fspcnz9R2LjjKNtuJ7iMS9wMQBlv53IUogdnVrHEeCBeWN5BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd62efbc8f8a668a6a1b494fd4b67a99972c8818cbe6c53d14ccd97ccdecb7a2","last_reissued_at":"2026-07-05T11:10:38.392550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:38.392550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating AI cyber capabilities with crowdsourced elicitation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Artem Petrov, Dmitrii Volkov","submitted_at":"2025-05-26T12:40:32Z","abstract_excerpt":"As AI systems become increasingly capable, understanding their offensive cyber potential is critical for informed governance and responsible deployment. However, it's hard to accurately bound their capabilities, and some prior evaluations dramatically underestimated them. The art of extracting maximum task-specific performance from AIs is called \"AI elicitation\", and today's safety organizations typically conduct it in-house. In this paper, we explore crowdsourcing elicitation efforts as an alternative to in-house elicitation work.\n  We host open-access AI tracks at two Capture The Flag (CTF) "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.19915","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.19915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.19915","created_at":"2026-07-05T11:10:38.392614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.19915v2","created_at":"2026-07-05T11:10:38.392614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.19915","created_at":"2026-07-05T11:10:38.392614+00:00"},{"alias_kind":"pith_short_12","alias_value":"7VRO7PEPRJTI","created_at":"2026-07-05T11:10:38.392614+00:00"},{"alias_kind":"pith_short_16","alias_value":"7VRO7PEPRJTIU2Q3","created_at":"2026-07-05T11:10:38.392614+00:00"},{"alias_kind":"pith_short_8","alias_value":"7VRO7PEP","created_at":"2026-07-05T11:10:38.392614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG","json":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG.json","graph_json":"https://pith.science/api/pith-number/7VRO7PEPRJTIU2Q3JFH5JNT2TG/graph.json","events_json":"https://pith.science/api/pith-number/7VRO7PEPRJTIU2Q3JFH5JNT2TG/events.json","paper":"https://pith.science/paper/7VRO7PEP"},"agent_actions":{"view_html":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG","download_json":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG.json","view_paper":"https://pith.science/paper/7VRO7PEP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.19915&json=true","fetch_graph":"https://pith.science/api/pith-number/7VRO7PEPRJTIU2Q3JFH5JNT2TG/graph.json","fetch_events":"https://pith.science/api/pith-number/7VRO7PEPRJTIU2Q3JFH5JNT2TG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG/action/storage_attestation","attest_author":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG/action/author_attestation","sign_citation":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG/action/citation_signature","submit_replication":"https://pith.science/pith/7VRO7PEPRJTIU2Q3JFH5JNT2TG/action/replication_record"}},"created_at":"2026-07-05T11:10:38.392614+00:00","updated_at":"2026-07-05T11:10:38.392614+00:00"}