{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MGOKLECJXDJY4TEFAUGYFPVS5H","short_pith_number":"pith:MGOKLECJ","schema_version":"1.0","canonical_sha256":"619ca59049b8d38e4c85050d82beb2e9ce2efbd0020296a0513737ebe8b82172","source":{"kind":"arxiv","id":"2410.22284","version":1},"attestation_state":"computed","paper":{"title":"Embedding-based classifiers can detect prompt injection attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Md. Ahsan Ayub, Subhabrata Majumdar","submitted_at":"2024-10-29T17:36:59Z","abstract_excerpt":"Large Language Models (LLMs) are seeing significant adoption in every type of organization due to their exceptional generative capabilities. However, LLMs are found to be vulnerable to various adversarial attacks, particularly prompt injection attacks, which trick them into producing harmful or inappropriate content. Adversaries execute such attacks by crafting malicious prompts to deceive the LLMs. In this paper, we propose a novel approach based on embedding-based Machine Learning (ML) classifiers to protect LLM-based applications against this severe threat. We leverage three commonly used e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.22284","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-10-29T17:36:59Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7121962ee74a23cc6088a2e50f0bf070319ca38cefbd69b2c5acd5e3ba42792e","abstract_canon_sha256":"029b199a2b2f907c14dcccfcb819ea3391799b2b961b38774aad75f5f19c2c6e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:06.986914Z","signature_b64":"eAiVcKxFu94qHLJLNYVtmIJTQg9zThu8PRcn7Vdtj7dnS9RljqvlqlbiDcr2pbGdsxaWLOkvuI9vxe1l0mImBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"619ca59049b8d38e4c85050d82beb2e9ce2efbd0020296a0513737ebe8b82172","last_reissued_at":"2026-07-05T09:28:06.986464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:06.986464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Embedding-based classifiers can detect prompt injection attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Md. Ahsan Ayub, Subhabrata Majumdar","submitted_at":"2024-10-29T17:36:59Z","abstract_excerpt":"Large Language Models (LLMs) are seeing significant adoption in every type of organization due to their exceptional generative capabilities. However, LLMs are found to be vulnerable to various adversarial attacks, particularly prompt injection attacks, which trick them into producing harmful or inappropriate content. Adversaries execute such attacks by crafting malicious prompts to deceive the LLMs. In this paper, we propose a novel approach based on embedding-based Machine Learning (ML) classifiers to protect LLM-based applications against this severe threat. We leverage three commonly used e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.22284","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.22284/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.22284","created_at":"2026-07-05T09:28:06.986523+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.22284v1","created_at":"2026-07-05T09:28:06.986523+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.22284","created_at":"2026-07-05T09:28:06.986523+00:00"},{"alias_kind":"pith_short_12","alias_value":"MGOKLECJXDJY","created_at":"2026-07-05T09:28:06.986523+00:00"},{"alias_kind":"pith_short_16","alias_value":"MGOKLECJXDJY4TEF","created_at":"2026-07-05T09:28:06.986523+00:00"},{"alias_kind":"pith_short_8","alias_value":"MGOKLECJ","created_at":"2026-07-05T09:28:06.986523+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15057","citing_title":"AutoDojo: Adaptive Black-Box Attacks Reveal the Limits of IPI Defenses and Task-Specification Effects in LLM Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03136","citing_title":"PsychoPass: Geometric Profiling of Multi-Turn Adversarial LLM Conversations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15842","citing_title":"Informationally Compressive Anonymization: Non-Degrading Sensitive Input Protection for Privacy-Preserving Supervised Machine Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27238","citing_title":"SafeTune: Mitigating Data Poisoning in LLM Fine-Tuning for RTL Code Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25562","citing_title":"SnapGuard: Lightweight Prompt Injection Detection for Screenshot-Based Web Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25716","citing_title":"Cross-Lingual Jailbreak Detection via Semantic Codebooks","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H","json":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H.json","graph_json":"https://pith.science/api/pith-number/MGOKLECJXDJY4TEFAUGYFPVS5H/graph.json","events_json":"https://pith.science/api/pith-number/MGOKLECJXDJY4TEFAUGYFPVS5H/events.json","paper":"https://pith.science/paper/MGOKLECJ"},"agent_actions":{"view_html":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H","download_json":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H.json","view_paper":"https://pith.science/paper/MGOKLECJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.22284&json=true","fetch_graph":"https://pith.science/api/pith-number/MGOKLECJXDJY4TEFAUGYFPVS5H/graph.json","fetch_events":"https://pith.science/api/pith-number/MGOKLECJXDJY4TEFAUGYFPVS5H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H/action/storage_attestation","attest_author":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H/action/author_attestation","sign_citation":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H/action/citation_signature","submit_replication":"https://pith.science/pith/MGOKLECJXDJY4TEFAUGYFPVS5H/action/replication_record"}},"created_at":"2026-07-05T09:28:06.986523+00:00","updated_at":"2026-07-05T09:28:06.986523+00:00"}