{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JHMHFF5RP7DN7ANYNXHN2RAW4Q","short_pith_number":"pith:JHMHFF5R","schema_version":"1.0","canonical_sha256":"49d87297b17fc6df81b86dcedd4416e4398953f5afcc75aed3a3c20f25f28710","source":{"kind":"arxiv","id":"2310.02949","version":1},"attestation_state":"computed","paper":{"title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dahua Lin, Linda Petzold, Qi Zhang, William Yang Wang, Xianjun Yang, Xiao Wang, Xun Zhao","submitted_at":"2023-10-04T16:39:31Z","abstract_excerpt":"Warning: This paper contains examples of harmful language, and reader discretion is recommended. The increasing open release of powerful large language models (LLMs) has facilitated the development of downstream applications by reducing the essential cost of data annotation and computation. To ensure AI safety, extensive safety-alignment measures have been conducted to armor these models against malicious use (primarily hard prompt attack). However, beneath the seemingly resilient facade of the armor, there might lurk a shadow. By simply tuning on 100 malicious examples with 1 GPU hour, these "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.02949","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-04T16:39:31Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"32267b5d46f9b61a065b3c5bb49538a40c7bf884871d6dd1ccf1ce3364f5e797","abstract_canon_sha256":"93581bc90d7e03c1a54b0c6e5a241a872150228feae6648637e1fde130501c3b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:57:15.818245Z","signature_b64":"EirjBWEvCa9rqP6R0KVOGzLXVyGF6M+irrgs7QjiRhXEROARA39oB8mrdEsZxuhKg4IX91GGYrjwyUDinE/xAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49d87297b17fc6df81b86dcedd4416e4398953f5afcc75aed3a3c20f25f28710","last_reissued_at":"2026-07-05T06:57:15.817757Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:57:15.817757Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dahua Lin, Linda Petzold, Qi Zhang, William Yang Wang, Xianjun Yang, Xiao Wang, Xun Zhao","submitted_at":"2023-10-04T16:39:31Z","abstract_excerpt":"Warning: This paper contains examples of harmful language, and reader discretion is recommended. The increasing open release of powerful large language models (LLMs) has facilitated the development of downstream applications by reducing the essential cost of data annotation and computation. To ensure AI safety, extensive safety-alignment measures have been conducted to armor these models against malicious use (primarily hard prompt attack). However, beneath the seemingly resilient facade of the armor, there might lurk a shadow. By simply tuning on 100 malicious examples with 1 GPU hour, these "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.02949","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.02949/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.02949","created_at":"2026-07-05T06:57:15.817813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.02949v1","created_at":"2026-07-05T06:57:15.817813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.02949","created_at":"2026-07-05T06:57:15.817813+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHMHFF5RP7DN","created_at":"2026-07-05T06:57:15.817813+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHMHFF5RP7DN7ANY","created_at":"2026-07-05T06:57:15.817813+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHMHFF5R","created_at":"2026-07-05T06:57:15.817813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":35,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22676","citing_title":"Skin-Deep: A Geometric Diagnostic for Alignment Fragility in Large Language Model Representations","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19168","citing_title":"Beyond Safe Data: Pretraining-Stage Alignment with Regular Safety Reflection","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15980","citing_title":"Do Activation Monitors Survive Model Updates? Benchmarking, Predicting, and Repairing Activation-Monitor Staleness","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12342","citing_title":"ALIGNBEAM : Inference-Time Alignment Transfer via Cross-Vocabulary Logit Mixing","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11316","citing_title":"Sch\\\"utzen: Evaluating LLM Safety in Bulgarian and German Contexts","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02111","citing_title":"Jailbreaking Multimodal Large Language Models using Multi-Clip Video","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01695","citing_title":"CANARY: Zero-Label Detection of Fine-Tuning Contamination in Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07631","citing_title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00160","citing_title":"DataShield: Safety-degrading Data Filtering for LLM Benign Instruction Fine-Tuning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31591","citing_title":"Evil Spectra: How Optimisers can Amplify or Suppress Emergent Misalignment","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14605","citing_title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24154","citing_title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30263","citing_title":"Defending Against Harmful Supervision Hidden in Benign Samples","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26526","citing_title":"Open-Weight LLM Fine-Tuning Defenses are Susceptible to Simple Attacks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28896","citing_title":"Feature Geometry of LoRA Adapters: A Sparse Autoencoder Analysis of Representational Divergence in Fine-Tuned Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28030","citing_title":"SPARD: Defending Harmful Fine-Tuning Attack via Safety Projection with Relevance-Diversity Data Selection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30640","citing_title":"CSULoRA: Closest Safe Update Low-Rank Adaptation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29396","citing_title":"Aligned but Fragile: Enhancing LLM Safety Robustness via Zeroth-Order Optimization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00557","citing_title":"Learning to Ask: When LLM Agents Meet Unclear Instruction","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":167,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21674","citing_title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16471","citing_title":"From AI-Generated Content to Agentic Action: Security and Safety Threats in Generative AI","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05367","citing_title":"Between a Rock and a Hard Place: The Tension Between Ethical Reasoning and Safety Alignment in LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08144","citing_title":"LLM Agents can Autonomously Exploit One-day Vulnerabilities","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q","json":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q.json","graph_json":"https://pith.science/api/pith-number/JHMHFF5RP7DN7ANYNXHN2RAW4Q/graph.json","events_json":"https://pith.science/api/pith-number/JHMHFF5RP7DN7ANYNXHN2RAW4Q/events.json","paper":"https://pith.science/paper/JHMHFF5R"},"agent_actions":{"view_html":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q","download_json":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q.json","view_paper":"https://pith.science/paper/JHMHFF5R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.02949&json=true","fetch_graph":"https://pith.science/api/pith-number/JHMHFF5RP7DN7ANYNXHN2RAW4Q/graph.json","fetch_events":"https://pith.science/api/pith-number/JHMHFF5RP7DN7ANYNXHN2RAW4Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q/action/storage_attestation","attest_author":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q/action/author_attestation","sign_citation":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q/action/citation_signature","submit_replication":"https://pith.science/pith/JHMHFF5RP7DN7ANYNXHN2RAW4Q/action/replication_record"}},"created_at":"2026-07-05T06:57:15.817813+00:00","updated_at":"2026-07-05T06:57:15.817813+00:00"}