{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AZK5LCCMVUXELF3IAHL5WMMUIM","short_pith_number":"pith:AZK5LCCM","schema_version":"1.0","canonical_sha256":"0655d5884cad2e45976801d7db31944338da84a6948d33503933d5730bac8081","source":{"kind":"arxiv","id":"2505.04806","version":2},"attestation_state":"computed","paper":{"title":"Red Teaming the Mind of the Machine: A Systematic Evaluation of Prompt Injection and Jailbreak Vulnerabilities in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Chetan Pathade","submitted_at":"2025-05-07T21:15:40Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly integrated into consumer and enterprise applications. Despite their capabilities, they remain susceptible to adversarial attacks such as prompt injection and jailbreaks that override alignment safeguards. This paper provides a systematic investigation of jailbreak strategies against various state-of-the-art LLMs. We categorize over 1,400 adversarial prompts, analyze their success against GPT-4, Claude 2, Mistral 7B, and Vicuna, and examine their generalizability and construction logic. We further propose layered mitigation strategies and recommend "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.04806","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-05-07T21:15:40Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"0f964dda37c7c4f7c65f643ca020d47a4457266645428137fc0af8f60f11c21b","abstract_canon_sha256":"c8adeecdde777acd6b073983a890b0596e5a4f1e70459357909613db2eb9b2bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:24.442684Z","signature_b64":"apwAR2iPay9DXZWkSSH4gmxDn665qD0j9lf5ENWCLtqyqMHnGTLG0fUuxM5ufD2CGUGDpjJoL86lUkYyVzivAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0655d5884cad2e45976801d7db31944338da84a6948d33503933d5730bac8081","last_reissued_at":"2026-07-05T11:02:24.442197Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:24.442197Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Red Teaming the Mind of the Machine: A Systematic Evaluation of Prompt Injection and Jailbreak Vulnerabilities in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Chetan Pathade","submitted_at":"2025-05-07T21:15:40Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly integrated into consumer and enterprise applications. Despite their capabilities, they remain susceptible to adversarial attacks such as prompt injection and jailbreaks that override alignment safeguards. This paper provides a systematic investigation of jailbreak strategies against various state-of-the-art LLMs. We categorize over 1,400 adversarial prompts, analyze their success against GPT-4, Claude 2, Mistral 7B, and Vicuna, and examine their generalizability and construction logic. We further propose layered mitigation strategies and recommend "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.04806","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.04806/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.04806","created_at":"2026-07-05T11:02:24.442254+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.04806v2","created_at":"2026-07-05T11:02:24.442254+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.04806","created_at":"2026-07-05T11:02:24.442254+00:00"},{"alias_kind":"pith_short_12","alias_value":"AZK5LCCMVUXE","created_at":"2026-07-05T11:02:24.442254+00:00"},{"alias_kind":"pith_short_16","alias_value":"AZK5LCCMVUXELF3I","created_at":"2026-07-05T11:02:24.442254+00:00"},{"alias_kind":"pith_short_8","alias_value":"AZK5LCCM","created_at":"2026-07-05T11:02:24.442254+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27027","citing_title":"ShareLock: A Stealthy Multi-Tool Threshold Poisoning Attack Against MCP","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02072","citing_title":"kNNGuard: Turning LLM Hidden Activations into a Training-Free Configurable Guardrail","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08876","citing_title":"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26999","citing_title":"Prompt Injection Detection is Regime-Dependent: A Deployment-Aware Evaluation with Interpretable Structural Signals","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30096","citing_title":"How Reliable Are AI Attackers Against a Fixed Vulnerable Target? A 400-Run Empirical Study of LLM Penetration Testing Consistency","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23723","citing_title":"MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22569","citing_title":"Whispers of Wealth: Red-Teaming Google's Agent Payments Protocol via Prompt Injection","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21110","citing_title":"Beyond Context: Large Language Models' Failure to Grasp Users' Intent","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08876","citing_title":"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19657","citing_title":"An AI Agent Execution Environment to Safeguard User Data","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM","json":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM.json","graph_json":"https://pith.science/api/pith-number/AZK5LCCMVUXELF3IAHL5WMMUIM/graph.json","events_json":"https://pith.science/api/pith-number/AZK5LCCMVUXELF3IAHL5WMMUIM/events.json","paper":"https://pith.science/paper/AZK5LCCM"},"agent_actions":{"view_html":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM","download_json":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM.json","view_paper":"https://pith.science/paper/AZK5LCCM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.04806&json=true","fetch_graph":"https://pith.science/api/pith-number/AZK5LCCMVUXELF3IAHL5WMMUIM/graph.json","fetch_events":"https://pith.science/api/pith-number/AZK5LCCMVUXELF3IAHL5WMMUIM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM/action/storage_attestation","attest_author":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM/action/author_attestation","sign_citation":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM/action/citation_signature","submit_replication":"https://pith.science/pith/AZK5LCCMVUXELF3IAHL5WMMUIM/action/replication_record"}},"created_at":"2026-07-05T11:02:24.442254+00:00","updated_at":"2026-07-05T11:02:24.442254+00:00"}