{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EQHDXP2DZPQFQPFQBOIV25B2OJ","short_pith_number":"pith:EQHDXP2D","schema_version":"1.0","canonical_sha256":"240e3bbf43cbe0583cb00b915d743a72645d7ce7982d2b2e5d10dae51953bebf","source":{"kind":"arxiv","id":"2502.00072","version":1},"attestation_state":"computed","paper":{"title":"LLM Cyber Evaluations Don't Capture Real-World Risk","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Adam Swanda, Kamil\\.e Luko\\v{s}i\\=ut\\.e","submitted_at":"2025-01-31T05:33:48Z","abstract_excerpt":"Large language models (LLMs) are demonstrating increasing prowess in cybersecurity applications, creating creating inherent risks alongside their potential for strengthening defenses. In this position paper, we argue that current efforts to evaluate risks posed by these capabilities are misaligned with the goal of understanding real-world impact. Evaluating LLM cybersecurity risk requires more than just measuring model capabilities -- it demands a comprehensive risk assessment that incorporates analysis of threat actor adoption behavior and potential for impact. We propose a risk assessment fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00072","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-01-31T05:33:48Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"31ba0d334f85b9bbe1016da0cf8c651ccf84421b56935f929ec1ff0d22322e5c","abstract_canon_sha256":"82735d0f90521b9b9ef21240f8bc3d0f038883ec0ffb7ffa2dfded088a8f9f7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:02.664662Z","signature_b64":"zcDFp7BB29zYzIJj1jOuXXG4nvRMfJlmWbwShUFu0DcSrvEZcU94QtWsLDq2G/0SEy2EUoXTTIG0lSm7pCQzBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"240e3bbf43cbe0583cb00b915d743a72645d7ce7982d2b2e5d10dae51953bebf","last_reissued_at":"2026-07-05T10:08:02.664181Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:02.664181Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Cyber Evaluations Don't Capture Real-World Risk","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Adam Swanda, Kamil\\.e Luko\\v{s}i\\=ut\\.e","submitted_at":"2025-01-31T05:33:48Z","abstract_excerpt":"Large language models (LLMs) are demonstrating increasing prowess in cybersecurity applications, creating creating inherent risks alongside their potential for strengthening defenses. In this position paper, we argue that current efforts to evaluate risks posed by these capabilities are misaligned with the goal of understanding real-world impact. Evaluating LLM cybersecurity risk requires more than just measuring model capabilities -- it demands a comprehensive risk assessment that incorporates analysis of threat actor adoption behavior and potential for impact. We propose a risk assessment fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00072","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00072","created_at":"2026-07-05T10:08:02.664240+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00072v1","created_at":"2026-07-05T10:08:02.664240+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00072","created_at":"2026-07-05T10:08:02.664240+00:00"},{"alias_kind":"pith_short_12","alias_value":"EQHDXP2DZPQF","created_at":"2026-07-05T10:08:02.664240+00:00"},{"alias_kind":"pith_short_16","alias_value":"EQHDXP2DZPQFQPFQ","created_at":"2026-07-05T10:08:02.664240+00:00"},{"alias_kind":"pith_short_8","alias_value":"EQHDXP2D","created_at":"2026-07-05T10:08:02.664240+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09809","citing_title":"Evaluation Cards: An Interpretive Layer for AI Evaluation Reporting","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10015","citing_title":"The coordination gap in frontier AI safety policies","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ","json":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ.json","graph_json":"https://pith.science/api/pith-number/EQHDXP2DZPQFQPFQBOIV25B2OJ/graph.json","events_json":"https://pith.science/api/pith-number/EQHDXP2DZPQFQPFQBOIV25B2OJ/events.json","paper":"https://pith.science/paper/EQHDXP2D"},"agent_actions":{"view_html":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ","download_json":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ.json","view_paper":"https://pith.science/paper/EQHDXP2D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00072&json=true","fetch_graph":"https://pith.science/api/pith-number/EQHDXP2DZPQFQPFQBOIV25B2OJ/graph.json","fetch_events":"https://pith.science/api/pith-number/EQHDXP2DZPQFQPFQBOIV25B2OJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ/action/storage_attestation","attest_author":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ/action/author_attestation","sign_citation":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ/action/citation_signature","submit_replication":"https://pith.science/pith/EQHDXP2DZPQFQPFQBOIV25B2OJ/action/replication_record"}},"created_at":"2026-07-05T10:08:02.664240+00:00","updated_at":"2026-07-05T10:08:02.664240+00:00"}