{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O7SQFOW7XOKUYFBVPYWHGAQTLJ","short_pith_number":"pith:O7SQFOW7","schema_version":"1.0","canonical_sha256":"77e502badfbb954c14357e2c7302135a4bdb5e854272acf21ac08d7dad9b1f5d","source":{"kind":"arxiv","id":"2507.06043","version":2},"attestation_state":"computed","paper":{"title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Jianhao Chen, Mayi Xu, Tieyun Qian, Xiaohu Li, Yunfeng Ning, Zepeng Bao","submitted_at":"2025-07-08T14:45:21Z","abstract_excerpt":"Security alignment enables the Large Language Model (LLM) to gain the protection against malicious queries, but various jailbreak attack methods reveal the vulnerability of this security mechanism. Previous studies have isolated LLM jailbreak attacks and defenses. We analyze the security protection mechanism of the LLM, and propose a framework that combines attack and defense. Our method is based on the linearly separable property of LLM intermediate layer embedding, as well as the essence of jailbreak attack, which aims to embed harmful problems and transfer them to the safe area. We utilize "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06043","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-07-08T14:45:21Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"595d92510be167272fc32045bb7230612dd4e94bdaaf1e427e4f6f94d68f1b7a","abstract_canon_sha256":"6530dae7980803ba77eee873b1f07d898cd22e5119206905043e3d39c6aee725"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:11.284604Z","signature_b64":"PXWIVVNCVmi4ap4lg5Fe1ccVgVcjl4eHfGL4wV4zw6rVxjv17yySjZ4TCOs+u1RZ12vSZH1oXuFPRkDHKPykBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77e502badfbb954c14357e2c7302135a4bdb5e854272acf21ac08d7dad9b1f5d","last_reissued_at":"2026-07-05T11:49:11.284057Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:11.284057Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Jianhao Chen, Mayi Xu, Tieyun Qian, Xiaohu Li, Yunfeng Ning, Zepeng Bao","submitted_at":"2025-07-08T14:45:21Z","abstract_excerpt":"Security alignment enables the Large Language Model (LLM) to gain the protection against malicious queries, but various jailbreak attack methods reveal the vulnerability of this security mechanism. Previous studies have isolated LLM jailbreak attacks and defenses. We analyze the security protection mechanism of the LLM, and propose a framework that combines attack and defense. Our method is based on the linearly separable property of LLM intermediate layer embedding, as well as the essence of jailbreak attack, which aims to embed harmful problems and transfer them to the safe area. We utilize "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06043","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06043","created_at":"2026-07-05T11:49:11.284115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06043v2","created_at":"2026-07-05T11:49:11.284115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06043","created_at":"2026-07-05T11:49:11.284115+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7SQFOW7XOKU","created_at":"2026-07-05T11:49:11.284115+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7SQFOW7XOKUYFBV","created_at":"2026-07-05T11:49:11.284115+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7SQFOW7","created_at":"2026-07-05T11:49:11.284115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.09473","citing_title":"NeuronTune: Fine-Grained Neuron Modulation for Balanced Safety-Utility Alignment in LLMs","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ","json":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ.json","graph_json":"https://pith.science/api/pith-number/O7SQFOW7XOKUYFBVPYWHGAQTLJ/graph.json","events_json":"https://pith.science/api/pith-number/O7SQFOW7XOKUYFBVPYWHGAQTLJ/events.json","paper":"https://pith.science/paper/O7SQFOW7"},"agent_actions":{"view_html":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ","download_json":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ.json","view_paper":"https://pith.science/paper/O7SQFOW7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06043&json=true","fetch_graph":"https://pith.science/api/pith-number/O7SQFOW7XOKUYFBVPYWHGAQTLJ/graph.json","fetch_events":"https://pith.science/api/pith-number/O7SQFOW7XOKUYFBVPYWHGAQTLJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ/action/storage_attestation","attest_author":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ/action/author_attestation","sign_citation":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ/action/citation_signature","submit_replication":"https://pith.science/pith/O7SQFOW7XOKUYFBVPYWHGAQTLJ/action/replication_record"}},"created_at":"2026-07-05T11:49:11.284115+00:00","updated_at":"2026-07-05T11:49:11.284115+00:00"}