{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VAN4SOPHW2OAPYGORNZGKLZDXX","short_pith_number":"pith:VAN4SOPH","schema_version":"1.0","canonical_sha256":"a81bc939e7b69c07e0ce8b72652f23bde8f823ce7ff7a40091d6524a38756714","source":{"kind":"arxiv","id":"2409.16165","version":3},"attestation_state":"computed","paper":{"title":"EnIGMA: Interactive Tools Substantially Assist LM Agents in Finding Security Vulnerabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Brendan Dolan-Gavitt, Carlos E. Jimenez, Farshad Khorrami, Haoran Xi, John Yang, Karthik Narasimhan, Kilian Lieret, Kimberly Milner, Meet Udeshi, Minghao Shao, Muhammad Shafique, Ofir Press, Prashanth Krishnamurthy, Ramesh Karri, Sofija Jancheska, Talor Abramovich","submitted_at":"2024-09-24T15:06:01Z","abstract_excerpt":"Although language model (LM) agents have demonstrated increased performance in multiple domains, including coding and web-browsing, their success in cybersecurity has been limited. We present EnIGMA, an LM agent for autonomously solving Capture The Flag (CTF) challenges. We introduce new tools and interfaces to improve the agent's ability to find and exploit security vulnerabilities, focusing on interactive terminal programs. These novel Interactive Agent Tools enable LM agents, for the first time, to run interactive utilities, such as a debugger and a server connection tool, which are essenti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.16165","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-09-24T15:06:01Z","cross_cats_sorted":[],"title_canon_sha256":"b15f18993bec78edfc083a4ced0317f43eb5a0d9ae8fe442742ee072fc0780c7","abstract_canon_sha256":"c19b2970787523c5d70a2fc0c234ac605705d58d09d6407886bcd1a8fcf474d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:29.581164Z","signature_b64":"WHtpTnTRgnQMY29T7xtmf8oCbUZjKQMhiFeQK/JK6RgnEnJmb9uejtqQKlre4F33x9WTWM94rmOw2GakqLQWDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a81bc939e7b69c07e0ce8b72652f23bde8f823ce7ff7a40091d6524a38756714","last_reissued_at":"2026-07-05T11:16:29.580359Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:29.580359Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EnIGMA: Interactive Tools Substantially Assist LM Agents in Finding Security Vulnerabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Brendan Dolan-Gavitt, Carlos E. Jimenez, Farshad Khorrami, Haoran Xi, John Yang, Karthik Narasimhan, Kilian Lieret, Kimberly Milner, Meet Udeshi, Minghao Shao, Muhammad Shafique, Ofir Press, Prashanth Krishnamurthy, Ramesh Karri, Sofija Jancheska, Talor Abramovich","submitted_at":"2024-09-24T15:06:01Z","abstract_excerpt":"Although language model (LM) agents have demonstrated increased performance in multiple domains, including coding and web-browsing, their success in cybersecurity has been limited. We present EnIGMA, an LM agent for autonomously solving Capture The Flag (CTF) challenges. We introduce new tools and interfaces to improve the agent's ability to find and exploit security vulnerabilities, focusing on interactive terminal programs. These novel Interactive Agent Tools enable LM agents, for the first time, to run interactive utilities, such as a debugger and a server connection tool, which are essenti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.16165","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.16165/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.16165","created_at":"2026-07-05T11:16:29.580463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.16165v3","created_at":"2026-07-05T11:16:29.580463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.16165","created_at":"2026-07-05T11:16:29.580463+00:00"},{"alias_kind":"pith_short_12","alias_value":"VAN4SOPHW2OA","created_at":"2026-07-05T11:16:29.580463+00:00"},{"alias_kind":"pith_short_16","alias_value":"VAN4SOPHW2OAPYGO","created_at":"2026-07-05T11:16:29.580463+00:00"},{"alias_kind":"pith_short_8","alias_value":"VAN4SOPH","created_at":"2026-07-05T11:16:29.580463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24402","citing_title":"Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22385","citing_title":"MetaPS: Adaptive Programmatic Strategy Selection for Market Agents","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01764","citing_title":"Mastermind: Strategy-grounded Learning for Repository-Scale Vulnerability Reproduction","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26195","citing_title":"CyberEvolver: Structured Self-Evolution for Cybersecurity Agents On the Fly","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29963","citing_title":"Honeyval: A Comprehensive Evaluation Framework for LLM-powered HTTP Honeypots","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15097","citing_title":"Veritas: Grounding LLM Agents for Reliable Vulnerability Reasoning over Stripped Binaries","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24212","citing_title":"Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06601","citing_title":"Patch2Vuln: Agentic Reconstruction of Vulnerabilities from Linux Distribution Binary Patches","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14317","citing_title":"Challenges and Future Directions in Agentic Reverse Engineering Systems","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17159","citing_title":"Systematic Capability Benchmarking of Frontier Large Language Models for Offensive Cyber Tasks","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20389","citing_title":"CyberCertBench: Evaluating LLMs in Cybersecurity Certification Knowledge","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX","json":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX.json","graph_json":"https://pith.science/api/pith-number/VAN4SOPHW2OAPYGORNZGKLZDXX/graph.json","events_json":"https://pith.science/api/pith-number/VAN4SOPHW2OAPYGORNZGKLZDXX/events.json","paper":"https://pith.science/paper/VAN4SOPH"},"agent_actions":{"view_html":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX","download_json":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX.json","view_paper":"https://pith.science/paper/VAN4SOPH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.16165&json=true","fetch_graph":"https://pith.science/api/pith-number/VAN4SOPHW2OAPYGORNZGKLZDXX/graph.json","fetch_events":"https://pith.science/api/pith-number/VAN4SOPHW2OAPYGORNZGKLZDXX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX/action/storage_attestation","attest_author":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX/action/author_attestation","sign_citation":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX/action/citation_signature","submit_replication":"https://pith.science/pith/VAN4SOPHW2OAPYGORNZGKLZDXX/action/replication_record"}},"created_at":"2026-07-05T11:16:29.580463+00:00","updated_at":"2026-07-05T11:16:29.580463+00:00"}