{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AV7W2TNT5R4TWZS6VJJLW6HY4D","short_pith_number":"pith:AV7W2TNT","schema_version":"1.0","canonical_sha256":"057f6d4db3ec793b665eaa52bb78f8e0eadb51e07d270a9d03042b8dc3b48407","source":{"kind":"arxiv","id":"2406.18510","version":1},"attestation_state":"computed","paper":{"title":"WildTeaming at Scale: From In-the-Wild Jailbreaks to (Adversarially) Safer Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Allyson Ettinger, Faeze Brahman, Kavel Rao, Liwei Jiang, Maarten Sap, Niloofar Mireshghallah, Nouha Dziri, Sachin Kumar, Seungju Han, Ximing Lu, Yejin Choi","submitted_at":"2024-06-26T17:31:22Z","abstract_excerpt":"We introduce WildTeaming, an automatic LLM safety red-teaming framework that mines in-the-wild user-chatbot interactions to discover 5.7K unique clusters of novel jailbreak tactics, and then composes multiple tactics for systematic exploration of novel jailbreaks. Compared to prior work that performed red-teaming via recruited human workers, gradient-based optimization, or iterative revision with LLMs, our work investigates jailbreaks from chatbot users who were not specifically instructed to break the system. WildTeaming reveals previously unidentified vulnerabilities of frontier LLMs, result"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18510","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-26T17:31:22Z","cross_cats_sorted":[],"title_canon_sha256":"3804d8e992a5c33f2a671ee1230e80209ff10e78054c08cd707046714c32ec77","abstract_canon_sha256":"2852bfbdb876798d549e42e386f090a21c5cdf4c7137c04e574236d213a7ca50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:10.308586Z","signature_b64":"ovr7db+XgPr53MqSI5s05EjjfeEbIbaUtcT1B0mYliUlfhPfIgr+E6GH6EkAJoLYElsG+4hzXPWpUNwb3y9eDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"057f6d4db3ec793b665eaa52bb78f8e0eadb51e07d270a9d03042b8dc3b48407","last_reissued_at":"2026-07-05T08:37:10.308113Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:10.308113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WildTeaming at Scale: From In-the-Wild Jailbreaks to (Adversarially) Safer Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Allyson Ettinger, Faeze Brahman, Kavel Rao, Liwei Jiang, Maarten Sap, Niloofar Mireshghallah, Nouha Dziri, Sachin Kumar, Seungju Han, Ximing Lu, Yejin Choi","submitted_at":"2024-06-26T17:31:22Z","abstract_excerpt":"We introduce WildTeaming, an automatic LLM safety red-teaming framework that mines in-the-wild user-chatbot interactions to discover 5.7K unique clusters of novel jailbreak tactics, and then composes multiple tactics for systematic exploration of novel jailbreaks. Compared to prior work that performed red-teaming via recruited human workers, gradient-based optimization, or iterative revision with LLMs, our work investigates jailbreaks from chatbot users who were not specifically instructed to break the system. WildTeaming reveals previously unidentified vulnerabilities of frontier LLMs, result"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18510","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18510/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18510","created_at":"2026-07-05T08:37:10.308174+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18510v1","created_at":"2026-07-05T08:37:10.308174+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18510","created_at":"2026-07-05T08:37:10.308174+00:00"},{"alias_kind":"pith_short_12","alias_value":"AV7W2TNT5R4T","created_at":"2026-07-05T08:37:10.308174+00:00"},{"alias_kind":"pith_short_16","alias_value":"AV7W2TNT5R4TWZS6","created_at":"2026-07-05T08:37:10.308174+00:00"},{"alias_kind":"pith_short_8","alias_value":"AV7W2TNT","created_at":"2026-07-05T08:37:10.308174+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07916","citing_title":"Persona Cartography: Charting Language Model Personality Traits in Weight Space","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25782","citing_title":"Do Encoders Suffice? A Systematic Comparison of Encoder and Decoder Safety Judges for LLM Adversarial Evaluation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02510","citing_title":"Online Safety Monitoring for LLMs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12342","citing_title":"ALIGNBEAM : Inference-Time Alignment Transfer via Cross-Vocabulary Logit Mixing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07678","citing_title":"DOG-DPO:Dynamic Optimization in Geometry for Safety Alignment","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03226","citing_title":"Self-Mined Hardness for Safety Fine-Tuning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26999","citing_title":"Prompt Injection Detection is Regime-Dependent: A Deployment-Aware Evaluation with Interpretable Structural Signals","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26396","citing_title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2504.00446","citing_title":"Exposing the Ghost in the Transformer: Abnormal Detection for Large Language Models via Hidden State Forensics","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2412.13663","citing_title":"Smarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21285","citing_title":"When Models Outthink Their Safety: Unveiling and Mitigating Self-Jailbreak in Large Reasoning Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17408","citing_title":"The Impact of Off-Policy Training Data on Probe Generalisation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2406.18495","citing_title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2410.18451","citing_title":"Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07655","citing_title":"Guardian-as-an-Advisor: Advancing Next-Generation Guardian Models for Trustworthy LLMs","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18473","citing_title":"Train Separately, Merge Together: Modular Post-Training with Mixture-of-Experts","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03226","citing_title":"Self-Mined Hardness for Safety Fine-Tuning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D","json":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D.json","graph_json":"https://pith.science/api/pith-number/AV7W2TNT5R4TWZS6VJJLW6HY4D/graph.json","events_json":"https://pith.science/api/pith-number/AV7W2TNT5R4TWZS6VJJLW6HY4D/events.json","paper":"https://pith.science/paper/AV7W2TNT"},"agent_actions":{"view_html":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D","download_json":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D.json","view_paper":"https://pith.science/paper/AV7W2TNT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18510&json=true","fetch_graph":"https://pith.science/api/pith-number/AV7W2TNT5R4TWZS6VJJLW6HY4D/graph.json","fetch_events":"https://pith.science/api/pith-number/AV7W2TNT5R4TWZS6VJJLW6HY4D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D/action/storage_attestation","attest_author":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D/action/author_attestation","sign_citation":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D/action/citation_signature","submit_replication":"https://pith.science/pith/AV7W2TNT5R4TWZS6VJJLW6HY4D/action/replication_record"}},"created_at":"2026-07-05T08:37:10.308174+00:00","updated_at":"2026-07-05T08:37:10.308174+00:00"}