{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DDOZ3WKPSOSD2JDLG72XGMXZLM","short_pith_number":"pith:DDOZ3WKP","schema_version":"1.0","canonical_sha256":"18dd9dd94f93a43d246b37f57332f95b0790c17fde9b0b11de7ae843e8915bf2","source":{"kind":"arxiv","id":"2310.06387","version":3},"attestation_state":"computed","paper":{"title":"Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Ang Li, Yichuan Mo, Yifei Wang, Yisen Wang, Zeming Wei","submitted_at":"2023-10-10T07:50:29Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable success in various tasks, yet their safety and the risk of generating harmful content remain pressing concerns. In this paper, we delve into the potential of In-Context Learning (ICL) to modulate the alignment of LLMs. Specifically, we propose the In-Context Attack (ICA) which employs harmful demonstrations to subvert LLMs, and the In-Context Defense (ICD) which bolsters model resilience through examples that demonstrate refusal to produce harmful responses. We offer theoretical insights to elucidate how a limited set of in-context demonstrati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06387","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-10T07:50:29Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CR"],"title_canon_sha256":"31f7bb9afcc76557695448e50fc0eccfce6ad703e4fbe6df55a883896e185eac","abstract_canon_sha256":"15dfd35c1215754c449dd13b3e56bd284060ef3239b8d06538899af3653532e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:12.845400Z","signature_b64":"iC8zzuO1OnKiyvnmC5fB8C6vMktkk12yScFOIsIhP4fXXJ2vHdmFoqR+5d+smEaDfIdFQmBjYApJs6kkaUwxBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18dd9dd94f93a43d246b37f57332f95b0790c17fde9b0b11de7ae843e8915bf2","last_reissued_at":"2026-07-05T08:23:12.844865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:12.844865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jailbreak and Guard Aligned Language Models with Only Few In-Context Demonstrations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Ang Li, Yichuan Mo, Yifei Wang, Yisen Wang, Zeming Wei","submitted_at":"2023-10-10T07:50:29Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable success in various tasks, yet their safety and the risk of generating harmful content remain pressing concerns. In this paper, we delve into the potential of In-Context Learning (ICL) to modulate the alignment of LLMs. Specifically, we propose the In-Context Attack (ICA) which employs harmful demonstrations to subvert LLMs, and the In-Context Defense (ICD) which bolsters model resilience through examples that demonstrate refusal to produce harmful responses. We offer theoretical insights to elucidate how a limited set of in-context demonstrati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06387","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06387/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06387","created_at":"2026-07-05T08:23:12.844925+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06387v3","created_at":"2026-07-05T08:23:12.844925+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06387","created_at":"2026-07-05T08:23:12.844925+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDOZ3WKPSOSD","created_at":"2026-07-05T08:23:12.844925+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDOZ3WKPSOSD2JDL","created_at":"2026-07-05T08:23:12.844925+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDOZ3WKP","created_at":"2026-07-05T08:23:12.844925+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07461","citing_title":"Mitigating Taint-Style Vulnerabilities in MCP Servers via Security-Aware Tool Descriptions","ref_index":53,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22237","citing_title":"Investigating The Security of Modern AI and Cloud Infrastructure","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19660","citing_title":"A Layered Security Framework Against Prompt Injection in RAG-Based Chatbots","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01738","citing_title":"THRD: A Training-Free Multi-Turn Defense Framework for Jailbreak Attacks on Large Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26409","citing_title":"Jailbreak susceptibility prediction and mitigation via the behavioral geometry of models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27981","citing_title":"ToxiREX: A Dataset on Toxic REasoning in ConteXt","ref_index":223,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00979","citing_title":"GradingAttack: Exposing Security Vulnerabilities in LLM Based Educational Grading Agents","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16737","citing_title":"Secure LLM Fine-Tuning via Safety-Aware Probing","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16551","citing_title":"PQR: A Framework to Generate Diverse and Realistic User Queries that Elicit QA Agent Failures","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01770","citing_title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09067","citing_title":"Enhancing the Safety of Medical Vision-Language Models by Synthetic Demonstrations","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20325","citing_title":"GUARD: Guideline Upholding Test through Adaptive Role-play and Jailbreak Diagnostics for LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20129","citing_title":"SAID: Safety-Aware Intent Defense via Prefix Probing for Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02356","citing_title":"ASTRA: An Automated Framework for Strategy Discovery, Retrieval, and Evolution for Jailbreaking LLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00822","citing_title":"ContextCov: Deriving and Enforcing Executable Constraints from Agent Instruction Files","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08496","citing_title":"Latent Personality Alignment: Improving Harmlessness Without Mentioning Harms","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23338","citing_title":"A Systematic Survey of Security Threats and Defenses in LLM-Based AI Agents: A Layered Attack Surface Framework","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22427","citing_title":"Automation-Exploit: A Multi-Agent LLM Framework for Adaptive Offensive Security with Digital Twin-Based Risk-Mitigated Exploitation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00123","citing_title":"Minimal, Local, Causal Explanations for Jailbreak Success in Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19638","citing_title":"SafetyALFRED: Evaluating Safety-Conscious Planning of Multimodal Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12232","citing_title":"TEMPLATEFUZZ: Fine-Grained Chat Template Fuzzing for Jailbreaking and Red Teaming LLMs","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07727","citing_title":"TrajGuard: Streaming Hidden-state Trajectory Detection for Decoding-time Jailbreak Defense","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09222","citing_title":"GRM: Utility-Aware Jailbreak Attacks on Audio LLMs via Gradient-Ratio Masking","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM","json":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM.json","graph_json":"https://pith.science/api/pith-number/DDOZ3WKPSOSD2JDLG72XGMXZLM/graph.json","events_json":"https://pith.science/api/pith-number/DDOZ3WKPSOSD2JDLG72XGMXZLM/events.json","paper":"https://pith.science/paper/DDOZ3WKP"},"agent_actions":{"view_html":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM","download_json":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM.json","view_paper":"https://pith.science/paper/DDOZ3WKP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06387&json=true","fetch_graph":"https://pith.science/api/pith-number/DDOZ3WKPSOSD2JDLG72XGMXZLM/graph.json","fetch_events":"https://pith.science/api/pith-number/DDOZ3WKPSOSD2JDLG72XGMXZLM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM/action/storage_attestation","attest_author":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM/action/author_attestation","sign_citation":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM/action/citation_signature","submit_replication":"https://pith.science/pith/DDOZ3WKPSOSD2JDLG72XGMXZLM/action/replication_record"}},"created_at":"2026-07-05T08:23:12.844925+00:00","updated_at":"2026-07-05T08:23:12.844925+00:00"}