{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AII7RK7IMEGTSXZ5GUK4ESFSYJ","short_pith_number":"pith:AII7RK7I","schema_version":"1.0","canonical_sha256":"0211f8abe8610d395f3d3515c248b2c27eb52602ce0bc32bbc228b1d8e859a83","source":{"kind":"arxiv","id":"2403.08424","version":2},"attestation_state":"computed","paper":{"title":"Distract Large Language Models for Automatic Jailbreak Attack","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Guanhua Chen, Yan Yang, Yun Chen, Zeguan Xiao","submitted_at":"2024-03-13T11:16:43Z","abstract_excerpt":"Extensive efforts have been made before the public release of Large language models (LLMs) to align their behaviors with human values. However, even meticulously aligned LLMs remain vulnerable to malicious manipulations such as jailbreaking, leading to unintended behaviors. In this work, we propose a novel black-box jailbreak framework for automated red teaming of LLMs. We designed malicious content concealing and memory reframing with an iterative optimization algorithm to jailbreak LLMs, motivated by the research about the distractibility and over-confidence phenomenon of LLMs. Extensive exp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08424","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-03-13T11:16:43Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"eea0569776142afcec339f29087bc6003f4dc746f4f27c1f9c590b0492f4e093","abstract_canon_sha256":"eef276a45d69ee2d2102e4d31eb95a8ec1e384ca816dc9bd8e1d15b88a8ae054"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:35.431021Z","signature_b64":"PtNgBG9vta2JZfpia/S/wMhXExKllJ03nFEmJSbVsVL9ueDYTc9+qmgHs6edR5O/XtFPxAKj+78qbZ1Z3OyQBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0211f8abe8610d395f3d3515c248b2c27eb52602ce0bc32bbc228b1d8e859a83","last_reissued_at":"2026-07-05T09:13:35.430549Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:35.430549Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distract Large Language Models for Automatic Jailbreak Attack","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Guanhua Chen, Yan Yang, Yun Chen, Zeguan Xiao","submitted_at":"2024-03-13T11:16:43Z","abstract_excerpt":"Extensive efforts have been made before the public release of Large language models (LLMs) to align their behaviors with human values. However, even meticulously aligned LLMs remain vulnerable to malicious manipulations such as jailbreaking, leading to unintended behaviors. In this work, we propose a novel black-box jailbreak framework for automated red teaming of LLMs. We designed malicious content concealing and memory reframing with an iterative optimization algorithm to jailbreak LLMs, motivated by the research about the distractibility and over-confidence phenomenon of LLMs. Extensive exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08424","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08424","created_at":"2026-07-05T09:13:35.430603+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08424v2","created_at":"2026-07-05T09:13:35.430603+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08424","created_at":"2026-07-05T09:13:35.430603+00:00"},{"alias_kind":"pith_short_12","alias_value":"AII7RK7IMEGT","created_at":"2026-07-05T09:13:35.430603+00:00"},{"alias_kind":"pith_short_16","alias_value":"AII7RK7IMEGTSXZ5","created_at":"2026-07-05T09:13:35.430603+00:00"},{"alias_kind":"pith_short_8","alias_value":"AII7RK7I","created_at":"2026-07-05T09:13:35.430603+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29237","citing_title":"Evolving Skill-Structured Attack Memory Enhances LLM Jailbreaking","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04204","citing_title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ","json":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ.json","graph_json":"https://pith.science/api/pith-number/AII7RK7IMEGTSXZ5GUK4ESFSYJ/graph.json","events_json":"https://pith.science/api/pith-number/AII7RK7IMEGTSXZ5GUK4ESFSYJ/events.json","paper":"https://pith.science/paper/AII7RK7I"},"agent_actions":{"view_html":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ","download_json":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ.json","view_paper":"https://pith.science/paper/AII7RK7I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08424&json=true","fetch_graph":"https://pith.science/api/pith-number/AII7RK7IMEGTSXZ5GUK4ESFSYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/AII7RK7IMEGTSXZ5GUK4ESFSYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ/action/storage_attestation","attest_author":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ/action/author_attestation","sign_citation":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ/action/citation_signature","submit_replication":"https://pith.science/pith/AII7RK7IMEGTSXZ5GUK4ESFSYJ/action/replication_record"}},"created_at":"2026-07-05T09:13:35.430603+00:00","updated_at":"2026-07-05T09:13:35.430603+00:00"}