{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:W5IOSD3D37HSFTDR7AKCHORBPI","short_pith_number":"pith:W5IOSD3D","schema_version":"1.0","canonical_sha256":"b750e90f63dfcf22cc71f81423ba217a1d4f5fbd74d8420d2c52db822f68e2ad","source":{"kind":"arxiv","id":"2306.15447","version":2},"attestation_state":"computed","paper":{"title":"Are aligned neural networks adversarially aligned?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anas Awadalla, Christopher A. Choquette-Choo, Daphne Ippolito, Florian Tramer, Irena Gao, Katherine Lee, Ludwig Schmidt, Matthew Jagielski, Milad Nasr, Nicholas Carlini, Pang Wei Koh","submitted_at":"2023-06-26T17:18:44Z","abstract_excerpt":"Large language models are now tuned to align with the goals of their creators, namely to be \"helpful and harmless.\" These models should respond helpfully to user questions, but refuse to answer requests that could cause harm. However, adversarial users can construct inputs which circumvent attempts at alignment. In this work, we study adversarial alignment, and ask to what extent these models remain aligned when interacting with an adversarial user who constructs worst-case inputs (adversarial examples). These inputs are designed to cause the model to emit harmful content that would otherwise "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.15447","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-26T17:18:44Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"3dac4d2aacd1a7c2a42264c63ae4f529b8a27f4528d32432da4b2ad2f9ce7b0c","abstract_canon_sha256":"5b84531df77c1f65770589a169c67333b1b68f51c214dc2f990f594c9720ea28"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:37.818380Z","signature_b64":"3S+OJ7A15wPPT2+YDiGHk/uLopZ1hZw2tIxoPUBERHrTDXsvOW0Piq7kfoYQvVSbO9Noh2xPMmTqSChPbiY1Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b750e90f63dfcf22cc71f81423ba217a1d4f5fbd74d8420d2c52db822f68e2ad","last_reissued_at":"2026-07-05T08:15:37.817894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:37.817894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are aligned neural networks adversarially aligned?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anas Awadalla, Christopher A. Choquette-Choo, Daphne Ippolito, Florian Tramer, Irena Gao, Katherine Lee, Ludwig Schmidt, Matthew Jagielski, Milad Nasr, Nicholas Carlini, Pang Wei Koh","submitted_at":"2023-06-26T17:18:44Z","abstract_excerpt":"Large language models are now tuned to align with the goals of their creators, namely to be \"helpful and harmless.\" These models should respond helpfully to user questions, but refuse to answer requests that could cause harm. However, adversarial users can construct inputs which circumvent attempts at alignment. In this work, we study adversarial alignment, and ask to what extent these models remain aligned when interacting with an adversarial user who constructs worst-case inputs (adversarial examples). These inputs are designed to cause the model to emit harmful content that would otherwise "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.15447","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.15447/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.15447","created_at":"2026-07-05T08:15:37.817950+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.15447v2","created_at":"2026-07-05T08:15:37.817950+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.15447","created_at":"2026-07-05T08:15:37.817950+00:00"},{"alias_kind":"pith_short_12","alias_value":"W5IOSD3D37HS","created_at":"2026-07-05T08:15:37.817950+00:00"},{"alias_kind":"pith_short_16","alias_value":"W5IOSD3D37HSFTDR","created_at":"2026-07-05T08:15:37.817950+00:00"},{"alias_kind":"pith_short_8","alias_value":"W5IOSD3D","created_at":"2026-07-05T08:15:37.817950+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07968","citing_title":"RecurGuard: Runtime Monitoring for Reasoning-Token Consumption Attacks","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2307.15043","citing_title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14113","citing_title":"Adversarial Hubness in Multi-Modal Retrieval","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21948","citing_title":"SCI-Defense: Defending Manipulation Attacks from Generative Engine Optimization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17310","citing_title":"Attention Hijacking: Response Manipulation Across Queries in Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06987","citing_title":"Catastrophic Jailbreak of Open-source LLMs via Exploiting Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2311.17035","citing_title":"Scalable Extraction of Training Data from (Production) Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03684","citing_title":"SmoothLLM: Defending Large Language Models Against Jailbreaking Attacks","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2309.00614","citing_title":"Baseline Defenses for Adversarial Attacks Against Aligned Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2310.03744","citing_title":"Improved Baselines with Visual Instruction Tuning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10582","citing_title":"Guaranteed Jailbreaking Defense via Disrupt-and-Rectify Smoothing","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04019","citing_title":"Redefining AI Red Teaming in the Agentic Era: From Weeks to Hours","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01449","citing_title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI","json":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI.json","graph_json":"https://pith.science/api/pith-number/W5IOSD3D37HSFTDR7AKCHORBPI/graph.json","events_json":"https://pith.science/api/pith-number/W5IOSD3D37HSFTDR7AKCHORBPI/events.json","paper":"https://pith.science/paper/W5IOSD3D"},"agent_actions":{"view_html":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI","download_json":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI.json","view_paper":"https://pith.science/paper/W5IOSD3D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.15447&json=true","fetch_graph":"https://pith.science/api/pith-number/W5IOSD3D37HSFTDR7AKCHORBPI/graph.json","fetch_events":"https://pith.science/api/pith-number/W5IOSD3D37HSFTDR7AKCHORBPI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI/action/storage_attestation","attest_author":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI/action/author_attestation","sign_citation":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI/action/citation_signature","submit_replication":"https://pith.science/pith/W5IOSD3D37HSFTDR7AKCHORBPI/action/replication_record"}},"created_at":"2026-07-05T08:15:37.817950+00:00","updated_at":"2026-07-05T08:15:37.817950+00:00"}