{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KWXXUUHVN4PWYPCSTL4AIG7ZP3","short_pith_number":"pith:KWXXUUHV","schema_version":"1.0","canonical_sha256":"55af7a50f56f1f6c3c529af8041bf97ed10fcf7f6387833d9bab8ab7e598735a","source":{"kind":"arxiv","id":"2502.05223","version":1},"attestation_state":"computed","paper":{"title":"KDA: A Knowledge-Distilled Attacker for Generating Diverse Prompts to Jailbreak LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Buyun Liang, Darshan Thaker, Jinqi Luo, Kwan Ho Ryan Chan, Ren\\'e Vidal","submitted_at":"2025-02-05T21:50:34Z","abstract_excerpt":"Jailbreak attacks exploit specific prompts to bypass LLM safeguards, causing the LLM to generate harmful, inappropriate, and misaligned content. Current jailbreaking methods rely heavily on carefully designed system prompts and numerous queries to achieve a single successful attack, which is costly and impractical for large-scale red-teaming. To address this challenge, we propose to distill the knowledge of an ensemble of SOTA attackers into a single open-source model, called Knowledge-Distilled Attacker (KDA), which is finetuned to automatically generate coherent and diverse attack prompts wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05223","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-02-05T21:50:34Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"ad2e7f86bbaa185a0221757ead825e70c34e5ffc20a97036302f5b932ccc3447","abstract_canon_sha256":"880bfda745cf82e9f0fdfc322ebe6056fce2cebab67d35606b56fba15912d5d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:34.349931Z","signature_b64":"cKcZgfrQCvaZ6Y92JdrUk6JJDGMVei0ylbFNjjgtW2LA29kAYZ9z4sDuRV9OyUAdp8ysgQ8mMs40pTCGTZHFCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"55af7a50f56f1f6c3c529af8041bf97ed10fcf7f6387833d9bab8ab7e598735a","last_reissued_at":"2026-07-05T10:11:34.349561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:34.349561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KDA: A Knowledge-Distilled Attacker for Generating Diverse Prompts to Jailbreak LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Buyun Liang, Darshan Thaker, Jinqi Luo, Kwan Ho Ryan Chan, Ren\\'e Vidal","submitted_at":"2025-02-05T21:50:34Z","abstract_excerpt":"Jailbreak attacks exploit specific prompts to bypass LLM safeguards, causing the LLM to generate harmful, inappropriate, and misaligned content. Current jailbreaking methods rely heavily on carefully designed system prompts and numerous queries to achieve a single successful attack, which is costly and impractical for large-scale red-teaming. To address this challenge, we propose to distill the knowledge of an ensemble of SOTA attackers into a single open-source model, called Knowledge-Distilled Attacker (KDA), which is finetuned to automatically generate coherent and diverse attack prompts wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05223","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05223/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05223","created_at":"2026-07-05T10:11:34.349610+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05223v1","created_at":"2026-07-05T10:11:34.349610+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05223","created_at":"2026-07-05T10:11:34.349610+00:00"},{"alias_kind":"pith_short_12","alias_value":"KWXXUUHVN4PW","created_at":"2026-07-05T10:11:34.349610+00:00"},{"alias_kind":"pith_short_16","alias_value":"KWXXUUHVN4PWYPCS","created_at":"2026-07-05T10:11:34.349610+00:00"},{"alias_kind":"pith_short_8","alias_value":"KWXXUUHV","created_at":"2026-07-05T10:11:34.349610+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07539","citing_title":"Prompt Governance? On Governing Technologies Governed by Natural Language","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11157","citing_title":"Response-Based Knowledge Distillation for Multilingual Jailbreak Prevention Unwittingly Compromises Safety","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":143,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3","json":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3.json","graph_json":"https://pith.science/api/pith-number/KWXXUUHVN4PWYPCSTL4AIG7ZP3/graph.json","events_json":"https://pith.science/api/pith-number/KWXXUUHVN4PWYPCSTL4AIG7ZP3/events.json","paper":"https://pith.science/paper/KWXXUUHV"},"agent_actions":{"view_html":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3","download_json":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3.json","view_paper":"https://pith.science/paper/KWXXUUHV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05223&json=true","fetch_graph":"https://pith.science/api/pith-number/KWXXUUHVN4PWYPCSTL4AIG7ZP3/graph.json","fetch_events":"https://pith.science/api/pith-number/KWXXUUHVN4PWYPCSTL4AIG7ZP3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3/action/storage_attestation","attest_author":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3/action/author_attestation","sign_citation":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3/action/citation_signature","submit_replication":"https://pith.science/pith/KWXXUUHVN4PWYPCSTL4AIG7ZP3/action/replication_record"}},"created_at":"2026-07-05T10:11:34.349610+00:00","updated_at":"2026-07-05T10:11:34.349610+00:00"}