{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6WL2VUAIID46UE4P3JSCOBLPYP","short_pith_number":"pith:6WL2VUAI","schema_version":"1.0","canonical_sha256":"f597aad00840f9ea138fda6427056fc3e431652a66131740dc2bb72ca8cc18a4","source":{"kind":"arxiv","id":"2410.09114","version":2},"attestation_state":"computed","paper":{"title":"Catastrophic Cyber Capabilities Benchmark (3CB): Robustly Evaluating LLM Agent Cyber Offense Capabilities","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"cs.CR","authors_text":"Andrey Anurin, Esben Kran, Jason Schreiber, Jonathan Ng, Kibo Schaffer","submitted_at":"2024-10-10T12:06:48Z","abstract_excerpt":"LLM agents have the potential to revolutionize defensive cyber operations, but their offensive capabilities are not yet fully understood. To prepare for emerging threats, model developers and governments are evaluating the cyber capabilities of foundation models. However, these assessments often lack transparency and a comprehensive focus on offensive capabilities. In response, we introduce the Catastrophic Cyber Capabilities Benchmark (3CB), a novel framework designed to rigorously assess the real-world offensive capabilities of LLM agents. Our evaluation of modern LLMs on 3CB reveals that fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09114","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CR","submitted_at":"2024-10-10T12:06:48Z","cross_cats_sorted":["cs.AI","cs.LG","cs.PF"],"title_canon_sha256":"fc0c09c293976b8792ec703cef86351dae02a36e611842b34af53763abf16a24","abstract_canon_sha256":"602888df618890dec04606b8e91775d60b5a3197d941c63f556f441819b80a48"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:22.086510Z","signature_b64":"MYb9bT6kSdGQVBuF141Gy0M03V8ib3Nk/otnclUt9vem49PZV9CrNw38K0k45/yjNxM3xi0kgQsat3WcQzF+DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f597aad00840f9ea138fda6427056fc3e431652a66131740dc2bb72ca8cc18a4","last_reissued_at":"2026-07-05T09:30:22.086143Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:22.086143Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Catastrophic Cyber Capabilities Benchmark (3CB): Robustly Evaluating LLM Agent Cyber Offense Capabilities","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"cs.CR","authors_text":"Andrey Anurin, Esben Kran, Jason Schreiber, Jonathan Ng, Kibo Schaffer","submitted_at":"2024-10-10T12:06:48Z","abstract_excerpt":"LLM agents have the potential to revolutionize defensive cyber operations, but their offensive capabilities are not yet fully understood. To prepare for emerging threats, model developers and governments are evaluating the cyber capabilities of foundation models. However, these assessments often lack transparency and a comprehensive focus on offensive capabilities. In response, we introduce the Catastrophic Cyber Capabilities Benchmark (3CB), a novel framework designed to rigorously assess the real-world offensive capabilities of LLM agents. Our evaluation of modern LLMs on 3CB reveals that fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09114","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09114","created_at":"2026-07-05T09:30:22.086204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09114v2","created_at":"2026-07-05T09:30:22.086204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09114","created_at":"2026-07-05T09:30:22.086204+00:00"},{"alias_kind":"pith_short_12","alias_value":"6WL2VUAIID46","created_at":"2026-07-05T09:30:22.086204+00:00"},{"alias_kind":"pith_short_16","alias_value":"6WL2VUAIID46UE4P","created_at":"2026-07-05T09:30:22.086204+00:00"},{"alias_kind":"pith_short_8","alias_value":"6WL2VUAI","created_at":"2026-07-05T09:30:22.086204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31755","citing_title":"A Technical Typology of AI Systems in Public Administration","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2410.09024","citing_title":"AgentHarm: A Benchmark for Measuring Harmfulness of LLM Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10597","citing_title":"CrackMeBench: Binary Reverse Engineering for Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01186","citing_title":"Trace: Unmasking AI Attack Agents Through Terminal Behavior Fingerprinting","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06019","citing_title":"CritBench: A Framework for Evaluating Cybersecurity Capabilities of Large Language Models in IEC 61850 Digital Substation Environments","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20389","citing_title":"CyberCertBench: Evaluating LLMs in Cybersecurity Certification Knowledge","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP","json":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP.json","graph_json":"https://pith.science/api/pith-number/6WL2VUAIID46UE4P3JSCOBLPYP/graph.json","events_json":"https://pith.science/api/pith-number/6WL2VUAIID46UE4P3JSCOBLPYP/events.json","paper":"https://pith.science/paper/6WL2VUAI"},"agent_actions":{"view_html":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP","download_json":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP.json","view_paper":"https://pith.science/paper/6WL2VUAI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09114&json=true","fetch_graph":"https://pith.science/api/pith-number/6WL2VUAIID46UE4P3JSCOBLPYP/graph.json","fetch_events":"https://pith.science/api/pith-number/6WL2VUAIID46UE4P3JSCOBLPYP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP/action/storage_attestation","attest_author":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP/action/author_attestation","sign_citation":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP/action/citation_signature","submit_replication":"https://pith.science/pith/6WL2VUAIID46UE4P3JSCOBLPYP/action/replication_record"}},"created_at":"2026-07-05T09:30:22.086204+00:00","updated_at":"2026-07-05T09:30:22.086204+00:00"}