{"work":{"id":"8fac4469-dd8b-4784-9ff6-13d2e74e57fb","openalex_id":"https://openalex.org/W4386270999","doi":"10.48550/arxiv.2308.14132","arxiv_id":"2308.14132","raw_key":null,"title":"Detecting Language Model Attacks with Perplexity","authors":null,"authors_text":"Gabriel Alon, Michael Kamfonas","year":2023,"venue":"cs.CL","abstract":"A novel hack involving Large Language Models (LLMs) has emerged, exploiting adversarial suffixes to deceive models into generating perilous responses. Such jailbreaks can trick LLMs into providing intricate instructions to a malicious user for creating explosives, orchestrating a bank heist, or facilitating the creation of offensive content. By evaluating the perplexity of queries with adversarial suffixes using an open-source LLM (GPT-2), we found that they have exceedingly high perplexity values. As we explored a broad range of regular (non-adversarial) prompt varieties, we concluded that false positives are a significant challenge for plain perplexity filtering. A Light-GBM trained on perplexity and token length resolved the false positives and correctly detected most adversarial attacks in the test set.","external_url":"https://arxiv.org/abs/2308.14132","cited_by_count":12,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2308.14132","created_at":"2026-05-09T06:15:38.395571+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Detecting Language Model Attacks with Perplexity","render_title":"Detecting Language Model Attacks with Perplexity"},"hub":{"state":{"work_id":"8fac4469-dd8b-4784-9ff6-13d2e74e57fb","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":45,"external_cited_by_count":12,"distinct_field_count":6,"first_pith_cited_at":"2023-10-05T17:01:53+00:00","last_pith_cited_at":"2026-06-23T21:14:53+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-23T07:09:40.744503+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":8},{"context_role":"baseline","n":3},{"context_role":"method","n":2}],"polarity_counts":[{"context_polarity":"background","n":8},{"context_polarity":"baseline","n":2},{"context_polarity":"use_method","n":2},{"context_polarity":"contest","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}