{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CIGSZSAVGT3Q2FB5N3XMRGQGTY","short_pith_number":"pith:CIGSZSAV","schema_version":"1.0","canonical_sha256":"120d2cc81534f70d143d6eeec89a069e3f4603d98621b797eece4acb5545e3b8","source":{"kind":"arxiv","id":"2502.15797","version":1},"attestation_state":"computed","paper":{"title":"OCCULT: Evaluating Large Language Models for Offensive Cyber Operation Capabilities","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Alex Byrne, Dan Martin, Ethan Michalak, Gianpaolo Russo, Guido Zarrella, Marissa Dotter, Michael Kouremetis, Michael Threet","submitted_at":"2025-02-18T19:33:14Z","abstract_excerpt":"The prospect of artificial intelligence (AI) competing in the adversarial landscape of cyber security has long been considered one of the most impactful, challenging, and potentially dangerous applications of AI. Here, we demonstrate a new approach to assessing AI's progress towards enabling and scaling real-world offensive cyber operations (OCO) tactics in use by modern threat actors. We detail OCCULT, a lightweight operational evaluation framework that allows cyber security experts to contribute to rigorous and repeatable measurement of the plausible cyber security risks associated with any "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.15797","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CR","submitted_at":"2025-02-18T19:33:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a0cc419b7d062289599c6f414aa2374b6c68d6562e3e93330c2875af48bd3f9a","abstract_canon_sha256":"799369997df1d9c9707d9db6990a9deea1c16d1a54d07a12000f62ac8b0e929c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:16.540590Z","signature_b64":"nVY+jrzegdsJhcmM+RYyOswJljtgs+sJ42VE4UzlSlhS9NrJjvPLxJdQtyHQQHFD+QVhVET745yEV38MkyyDBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"120d2cc81534f70d143d6eeec89a069e3f4603d98621b797eece4acb5545e3b8","last_reissued_at":"2026-07-05T10:18:16.540180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:16.540180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OCCULT: Evaluating Large Language Models for Offensive Cyber Operation Capabilities","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Alex Byrne, Dan Martin, Ethan Michalak, Gianpaolo Russo, Guido Zarrella, Marissa Dotter, Michael Kouremetis, Michael Threet","submitted_at":"2025-02-18T19:33:14Z","abstract_excerpt":"The prospect of artificial intelligence (AI) competing in the adversarial landscape of cyber security has long been considered one of the most impactful, challenging, and potentially dangerous applications of AI. Here, we demonstrate a new approach to assessing AI's progress towards enabling and scaling real-world offensive cyber operations (OCO) tactics in use by modern threat actors. We detail OCCULT, a lightweight operational evaluation framework that allows cyber security experts to contribute to rigorous and repeatable measurement of the plausible cyber security risks associated with any "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15797","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15797/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.15797","created_at":"2026-07-05T10:18:16.540227+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.15797v1","created_at":"2026-07-05T10:18:16.540227+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15797","created_at":"2026-07-05T10:18:16.540227+00:00"},{"alias_kind":"pith_short_12","alias_value":"CIGSZSAVGT3Q","created_at":"2026-07-05T10:18:16.540227+00:00"},{"alias_kind":"pith_short_16","alias_value":"CIGSZSAVGT3Q2FB5","created_at":"2026-07-05T10:18:16.540227+00:00"},{"alias_kind":"pith_short_8","alias_value":"CIGSZSAV","created_at":"2026-07-05T10:18:16.540227+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.17753","citing_title":"The 2025 AI Agent Index: Documenting Technical and Safety Features of Deployed Agentic AI Systems","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19678","citing_title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17159","citing_title":"Systematic Capability Benchmarking of Frontier Large Language Models for Offensive Cyber Tasks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20389","citing_title":"CyberCertBench: Evaluating LLMs in Cybersecurity Certification Knowledge","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY","json":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY.json","graph_json":"https://pith.science/api/pith-number/CIGSZSAVGT3Q2FB5N3XMRGQGTY/graph.json","events_json":"https://pith.science/api/pith-number/CIGSZSAVGT3Q2FB5N3XMRGQGTY/events.json","paper":"https://pith.science/paper/CIGSZSAV"},"agent_actions":{"view_html":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY","download_json":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY.json","view_paper":"https://pith.science/paper/CIGSZSAV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.15797&json=true","fetch_graph":"https://pith.science/api/pith-number/CIGSZSAVGT3Q2FB5N3XMRGQGTY/graph.json","fetch_events":"https://pith.science/api/pith-number/CIGSZSAVGT3Q2FB5N3XMRGQGTY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY/action/storage_attestation","attest_author":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY/action/author_attestation","sign_citation":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY/action/citation_signature","submit_replication":"https://pith.science/pith/CIGSZSAVGT3Q2FB5N3XMRGQGTY/action/replication_record"}},"created_at":"2026-07-05T10:18:16.540227+00:00","updated_at":"2026-07-05T10:18:16.540227+00:00"}