{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AHZF3RFSZ6BQCERNLYY66WEM7J","short_pith_number":"pith:AHZF3RFS","schema_version":"1.0","canonical_sha256":"01f25dc4b2cf8301122d5e31ef588cfa69b68c5b852783a8e450acfa11c613ff","source":{"kind":"arxiv","id":"2408.04811","version":4},"attestation_state":"computed","paper":{"title":"h4rm3l: A language for Composable Jailbreak Attack Synthesis","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ananjan Nandi, Anna Goldie, Christopher D. Manning, Dan Jurafsky, Davide Ghilardi, Federico Bianchi, Gabriel Poesia, Moussa Koulako Bala Doumbouya","submitted_at":"2024-08-09T01:45:39Z","abstract_excerpt":"Despite their demonstrated valuable capabilities, state-of-the-art (SOTA) widely deployed large language models (LLMs) still have the potential to cause harm to society due to the ineffectiveness of their safety filters, which can be bypassed by prompt transformations called jailbreak attacks. Current approaches to LLM safety assessment, which employ datasets of templated prompts and benchmarking pipelines, fail to cover sufficiently large and diverse sets of jailbreak attacks, leading to the widespread deployment of unsafe LLMs. Recent research showed that novel jailbreak attacks could be der"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.04811","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CR","submitted_at":"2024-08-09T01:45:39Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY","cs.LG"],"title_canon_sha256":"bfab6943f905ba59c3d64b99b323c50d1a90d8e27db06a2472067b95aa75c0b1","abstract_canon_sha256":"0495108ba4e238d401e838f969d23cfdac9c55c6a311e00414517d41afba77fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:35.148551Z","signature_b64":"QpidqN2jK4iO5FiQqKvYC0iT0dvpe3O1wCrnnzHCieeNqbznLyQgbKgbGhBaITAT/O/PHGO/BWqjDMc4VYwYBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"01f25dc4b2cf8301122d5e31ef588cfa69b68c5b852783a8e450acfa11c613ff","last_reissued_at":"2026-07-05T10:38:35.148066Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:35.148066Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"h4rm3l: A language for Composable Jailbreak Attack Synthesis","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ananjan Nandi, Anna Goldie, Christopher D. Manning, Dan Jurafsky, Davide Ghilardi, Federico Bianchi, Gabriel Poesia, Moussa Koulako Bala Doumbouya","submitted_at":"2024-08-09T01:45:39Z","abstract_excerpt":"Despite their demonstrated valuable capabilities, state-of-the-art (SOTA) widely deployed large language models (LLMs) still have the potential to cause harm to society due to the ineffectiveness of their safety filters, which can be bypassed by prompt transformations called jailbreak attacks. Current approaches to LLM safety assessment, which employ datasets of templated prompts and benchmarking pipelines, fail to cover sufficiently large and diverse sets of jailbreak attacks, leading to the widespread deployment of unsafe LLMs. Recent research showed that novel jailbreak attacks could be der"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.04811","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.04811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.04811","created_at":"2026-07-05T10:38:35.148121+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.04811v4","created_at":"2026-07-05T10:38:35.148121+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.04811","created_at":"2026-07-05T10:38:35.148121+00:00"},{"alias_kind":"pith_short_12","alias_value":"AHZF3RFSZ6BQ","created_at":"2026-07-05T10:38:35.148121+00:00"},{"alias_kind":"pith_short_16","alias_value":"AHZF3RFSZ6BQCERN","created_at":"2026-07-05T10:38:35.148121+00:00"},{"alias_kind":"pith_short_8","alias_value":"AHZF3RFS","created_at":"2026-07-05T10:38:35.148121+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18193","citing_title":"A Red-Team Study of Anthropic Fable 5 & Opus 4.8 Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12710","citing_title":"Evolve the Method, Not the Prompts: Evolutionary Synthesis of Jailbreak Attacks on LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20351","citing_title":"Refusal Evaluation in Coding LLMs and Code Agents: A Systematic Review of Thirteen Malicious-Code Prompt Corpora (2023-2025)","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17614","citing_title":"Characterizing Model-Native Skills","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J","json":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J.json","graph_json":"https://pith.science/api/pith-number/AHZF3RFSZ6BQCERNLYY66WEM7J/graph.json","events_json":"https://pith.science/api/pith-number/AHZF3RFSZ6BQCERNLYY66WEM7J/events.json","paper":"https://pith.science/paper/AHZF3RFS"},"agent_actions":{"view_html":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J","download_json":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J.json","view_paper":"https://pith.science/paper/AHZF3RFS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.04811&json=true","fetch_graph":"https://pith.science/api/pith-number/AHZF3RFSZ6BQCERNLYY66WEM7J/graph.json","fetch_events":"https://pith.science/api/pith-number/AHZF3RFSZ6BQCERNLYY66WEM7J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J/action/storage_attestation","attest_author":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J/action/author_attestation","sign_citation":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J/action/citation_signature","submit_replication":"https://pith.science/pith/AHZF3RFSZ6BQCERNLYY66WEM7J/action/replication_record"}},"created_at":"2026-07-05T10:38:35.148121+00:00","updated_at":"2026-07-05T10:38:35.148121+00:00"}