{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GGPJ3P3B6DN3FXU4CMVJI2F6XD","short_pith_number":"pith:GGPJ3P3B","schema_version":"1.0","canonical_sha256":"319e9dbf61f0dbb2de9c132a9468beb8e205d083b27f06c2586d9ddecd93c080","source":{"kind":"arxiv","id":"2412.15623","version":1},"attestation_state":"computed","paper":{"title":"JailPO: A Novel Black-box Jailbreak Framework via Preference Optimization against Aligned LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Chu Wang, Hongyi Li, Jiawei Ye, Jie Wu, Tianjie Yan, Zhixin Li","submitted_at":"2024-12-20T07:29:10Z","abstract_excerpt":"Large Language Models (LLMs) aligned with human feedback have recently garnered significant attention. However, it remains vulnerable to jailbreak attacks, where adversaries manipulate prompts to induce harmful outputs. Exploring jailbreak attacks enables us to investigate the vulnerabilities of LLMs and further guides us in enhancing their security. Unfortunately, existing techniques mainly rely on handcrafted templates or generated-based optimization, posing challenges in scalability, efficiency and universality. To address these issues, we present JailPO, a novel black-box jailbreak framewo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15623","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-12-20T07:29:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bee320a62c6873a0a8fc6ba5088ca646a3a481c10c3ae5ad138864402d5c972d","abstract_canon_sha256":"aff95a802b3fcc5c524b229643e1d235071b4911b05dcc59c6c2208d5fbbe593"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:26.963218Z","signature_b64":"EQQCWHtjnCjAES2vbUrdGTrVS0LD7bZCWQYWdrAJNQanC2HscjONjlQsafDD5x/1WHwAUdWmfvbyANX43nJhDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"319e9dbf61f0dbb2de9c132a9468beb8e205d083b27f06c2586d9ddecd93c080","last_reissued_at":"2026-07-05T09:52:26.962761Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:26.962761Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JailPO: A Novel Black-box Jailbreak Framework via Preference Optimization against Aligned LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Chu Wang, Hongyi Li, Jiawei Ye, Jie Wu, Tianjie Yan, Zhixin Li","submitted_at":"2024-12-20T07:29:10Z","abstract_excerpt":"Large Language Models (LLMs) aligned with human feedback have recently garnered significant attention. However, it remains vulnerable to jailbreak attacks, where adversaries manipulate prompts to induce harmful outputs. Exploring jailbreak attacks enables us to investigate the vulnerabilities of LLMs and further guides us in enhancing their security. Unfortunately, existing techniques mainly rely on handcrafted templates or generated-based optimization, posing challenges in scalability, efficiency and universality. To address these issues, we present JailPO, a novel black-box jailbreak framewo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15623","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15623/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15623","created_at":"2026-07-05T09:52:26.962818+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15623v1","created_at":"2026-07-05T09:52:26.962818+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15623","created_at":"2026-07-05T09:52:26.962818+00:00"},{"alias_kind":"pith_short_12","alias_value":"GGPJ3P3B6DN3","created_at":"2026-07-05T09:52:26.962818+00:00"},{"alias_kind":"pith_short_16","alias_value":"GGPJ3P3B6DN3FXU4","created_at":"2026-07-05T09:52:26.962818+00:00"},{"alias_kind":"pith_short_8","alias_value":"GGPJ3P3B","created_at":"2026-07-05T09:52:26.962818+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29237","citing_title":"Evolving Skill-Structured Attack Memory Enhances LLM Jailbreaking","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD","json":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD.json","graph_json":"https://pith.science/api/pith-number/GGPJ3P3B6DN3FXU4CMVJI2F6XD/graph.json","events_json":"https://pith.science/api/pith-number/GGPJ3P3B6DN3FXU4CMVJI2F6XD/events.json","paper":"https://pith.science/paper/GGPJ3P3B"},"agent_actions":{"view_html":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD","download_json":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD.json","view_paper":"https://pith.science/paper/GGPJ3P3B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15623&json=true","fetch_graph":"https://pith.science/api/pith-number/GGPJ3P3B6DN3FXU4CMVJI2F6XD/graph.json","fetch_events":"https://pith.science/api/pith-number/GGPJ3P3B6DN3FXU4CMVJI2F6XD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD/action/storage_attestation","attest_author":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD/action/author_attestation","sign_citation":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD/action/citation_signature","submit_replication":"https://pith.science/pith/GGPJ3P3B6DN3FXU4CMVJI2F6XD/action/replication_record"}},"created_at":"2026-07-05T09:52:26.962818+00:00","updated_at":"2026-07-05T09:52:26.962818+00:00"}