{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JJBQJMR2S5BQTZYNZ4TUI3JMX7","short_pith_number":"pith:JJBQJMR2","schema_version":"1.0","canonical_sha256":"4a4304b23a974309e70dcf27446d2cbff69b5d9d301238d2cc2510f217b15c1d","source":{"kind":"arxiv","id":"2402.14872","version":2},"attestation_state":"computed","paper":{"title":"Semantic Mirror Jailbreak: Genetic Algorithm Based Jailbreak Prompts Against Open-source LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.CL","authors_text":"Aishan Liu, Ee-Chien Chang, Han Fang, Jiyi Zhang, Siyuan Liang, Xiaoxia Li","submitted_at":"2024-02-21T15:13:50Z","abstract_excerpt":"Large Language Models (LLMs), used in creative writing, code generation, and translation, generate text based on input sequences but are vulnerable to jailbreak attacks, where crafted prompts induce harmful outputs. Most jailbreak prompt methods use a combination of jailbreak templates followed by questions to ask to create jailbreak prompts. However, existing jailbreak prompt designs generally suffer from excessive semantic differences, resulting in an inability to resist defenses that use simple semantic metrics as thresholds. Jailbreak prompts are semantically more varied than the original "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14872","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T15:13:50Z","cross_cats_sorted":["cs.AI","cs.NE"],"title_canon_sha256":"30b3a078ee7db3d9acc8d6c2df44f7c286f6583eb9c0fa9460db348abc8dfa95","abstract_canon_sha256":"98c2ec71abbece9eafa4f12f132ff8ce2ac4db7880fe5e7026fb4d555c8349bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:37.535188Z","signature_b64":"iPbXUy53TSmftU6VMPsShWyqxPNDC9DOtwDLa4nGJxc0e6FKrCp/Mr3GdNUdCGUVQ1kHm2kfDdVhNBdffIGvAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a4304b23a974309e70dcf27446d2cbff69b5d9d301238d2cc2510f217b15c1d","last_reissued_at":"2026-07-05T07:49:37.534730Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:37.534730Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Semantic Mirror Jailbreak: Genetic Algorithm Based Jailbreak Prompts Against Open-source LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.CL","authors_text":"Aishan Liu, Ee-Chien Chang, Han Fang, Jiyi Zhang, Siyuan Liang, Xiaoxia Li","submitted_at":"2024-02-21T15:13:50Z","abstract_excerpt":"Large Language Models (LLMs), used in creative writing, code generation, and translation, generate text based on input sequences but are vulnerable to jailbreak attacks, where crafted prompts induce harmful outputs. Most jailbreak prompt methods use a combination of jailbreak templates followed by questions to ask to create jailbreak prompts. However, existing jailbreak prompt designs generally suffer from excessive semantic differences, resulting in an inability to resist defenses that use simple semantic metrics as thresholds. Jailbreak prompts are semantically more varied than the original "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14872","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14872","created_at":"2026-07-05T07:49:37.534786+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14872v2","created_at":"2026-07-05T07:49:37.534786+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14872","created_at":"2026-07-05T07:49:37.534786+00:00"},{"alias_kind":"pith_short_12","alias_value":"JJBQJMR2S5BQ","created_at":"2026-07-05T07:49:37.534786+00:00"},{"alias_kind":"pith_short_16","alias_value":"JJBQJMR2S5BQTZYN","created_at":"2026-07-05T07:49:37.534786+00:00"},{"alias_kind":"pith_short_8","alias_value":"JJBQJMR2","created_at":"2026-07-05T07:49:37.534786+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05783","citing_title":"Benchmarking the Robustness of Autonomous Driving to Environmental Illusions: A Lane Perception Perspective","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2601.04740","citing_title":"StealthGraph: Exposing Domain-Specific Risks in LLMs through Knowledge-Graph-Guided Harmful Prompt Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05116","citing_title":"On the Hardness of Junking LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10326","citing_title":"Jailbreaking the Matrix: Nullspace Steering for Controlled Model Subversion","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04630","citing_title":"Multimodal Backdoor Attack on VLMs for Autonomous Driving via Graffiti and Cross-Lingual Triggers","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7","json":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7.json","graph_json":"https://pith.science/api/pith-number/JJBQJMR2S5BQTZYNZ4TUI3JMX7/graph.json","events_json":"https://pith.science/api/pith-number/JJBQJMR2S5BQTZYNZ4TUI3JMX7/events.json","paper":"https://pith.science/paper/JJBQJMR2"},"agent_actions":{"view_html":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7","download_json":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7.json","view_paper":"https://pith.science/paper/JJBQJMR2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14872&json=true","fetch_graph":"https://pith.science/api/pith-number/JJBQJMR2S5BQTZYNZ4TUI3JMX7/graph.json","fetch_events":"https://pith.science/api/pith-number/JJBQJMR2S5BQTZYNZ4TUI3JMX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7/action/storage_attestation","attest_author":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7/action/author_attestation","sign_citation":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7/action/citation_signature","submit_replication":"https://pith.science/pith/JJBQJMR2S5BQTZYNZ4TUI3JMX7/action/replication_record"}},"created_at":"2026-07-05T07:49:37.534786+00:00","updated_at":"2026-07-05T07:49:37.534786+00:00"}