{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5VPCW45YR63YFQKSH5SC2HVYXH","short_pith_number":"pith:5VPCW45Y","schema_version":"1.0","canonical_sha256":"ed5e2b73b88fb782c1523f642d1eb8b9d2af6dfcc019a89c2d562cc5e3a52814","source":{"kind":"arxiv","id":"2408.01420","version":1},"attestation_state":"computed","paper":{"title":"Mission Impossible: A Statistical Perspective on Jailbreaking LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingtong Su, Julia Kempe, Karen Ullrich","submitted_at":"2024-08-02T17:55:50Z","abstract_excerpt":"Large language models (LLMs) are trained on a deluge of text data with limited quality control. As a result, LLMs can exhibit unintended or even harmful behaviours, such as leaking information, fake news or hate speech. Countermeasures, commonly referred to as preference alignment, include fine-tuning the pretrained LLMs with carefully crafted text examples of desired behaviour. Even then, empirical evidence shows preference aligned LLMs can be enticed to harmful behaviour. This so called jailbreaking of LLMs is typically achieved by adversarially modifying the input prompt to the LLM. Our pap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.01420","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-02T17:55:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d7481317065b5b09cbc1785ebb99a29e518f115fa3f49ee5c9eeb29538b41d9f","abstract_canon_sha256":"73956068add112ff32f6480892a38ce1e13edd10bcd1eef49c972e13dddee5d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:38.203568Z","signature_b64":"C9zLBX7w26pD+zFnAuCeWU82U04GZEBE4xbhpkw8wzfpuYBHbRmCTHkxn45LuNdDHeT7E/b3ADYV6NPsO6OBCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed5e2b73b88fb782c1523f642d1eb8b9d2af6dfcc019a89c2d562cc5e3a52814","last_reissued_at":"2026-07-05T08:51:38.203018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:38.203018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mission Impossible: A Statistical Perspective on Jailbreaking LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jingtong Su, Julia Kempe, Karen Ullrich","submitted_at":"2024-08-02T17:55:50Z","abstract_excerpt":"Large language models (LLMs) are trained on a deluge of text data with limited quality control. As a result, LLMs can exhibit unintended or even harmful behaviours, such as leaking information, fake news or hate speech. Countermeasures, commonly referred to as preference alignment, include fine-tuning the pretrained LLMs with carefully crafted text examples of desired behaviour. Even then, empirical evidence shows preference aligned LLMs can be enticed to harmful behaviour. This so called jailbreaking of LLMs is typically achieved by adversarially modifying the input prompt to the LLM. Our pap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.01420","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.01420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.01420","created_at":"2026-07-05T08:51:38.203075+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.01420v1","created_at":"2026-07-05T08:51:38.203075+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.01420","created_at":"2026-07-05T08:51:38.203075+00:00"},{"alias_kind":"pith_short_12","alias_value":"5VPCW45YR63Y","created_at":"2026-07-05T08:51:38.203075+00:00"},{"alias_kind":"pith_short_16","alias_value":"5VPCW45YR63YFQKS","created_at":"2026-07-05T08:51:38.203075+00:00"},{"alias_kind":"pith_short_8","alias_value":"5VPCW45Y","created_at":"2026-07-05T08:51:38.203075+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH","json":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH.json","graph_json":"https://pith.science/api/pith-number/5VPCW45YR63YFQKSH5SC2HVYXH/graph.json","events_json":"https://pith.science/api/pith-number/5VPCW45YR63YFQKSH5SC2HVYXH/events.json","paper":"https://pith.science/paper/5VPCW45Y"},"agent_actions":{"view_html":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH","download_json":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH.json","view_paper":"https://pith.science/paper/5VPCW45Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.01420&json=true","fetch_graph":"https://pith.science/api/pith-number/5VPCW45YR63YFQKSH5SC2HVYXH/graph.json","fetch_events":"https://pith.science/api/pith-number/5VPCW45YR63YFQKSH5SC2HVYXH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH/action/storage_attestation","attest_author":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH/action/author_attestation","sign_citation":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH/action/citation_signature","submit_replication":"https://pith.science/pith/5VPCW45YR63YFQKSH5SC2HVYXH/action/replication_record"}},"created_at":"2026-07-05T08:51:38.203075+00:00","updated_at":"2026-07-05T08:51:38.203075+00:00"}