{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IU24EYDIS34XNDQMHLVORY564T","short_pith_number":"pith:IU24EYDI","schema_version":"1.0","canonical_sha256":"4535c2606896f9768e0c3aeae8e3bee4e9bc61b8b07c389a2a4e6716e3d18f9b","source":{"kind":"arxiv","id":"2508.20038","version":3},"attestation_state":"computed","paper":{"title":"Forewarned is Forearmed: Pre-Synthesizing Jailbreak-like Instructions to Enhance LLM Safety Guardrail to Potential Attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Danding Wang, Guang Yang, Juan Cao, Qiang Sheng, Sheng Liu, Yang Li","submitted_at":"2025-08-27T16:44:03Z","abstract_excerpt":"Despite advances in improving large language model (LLM) to refuse to answer malicious instructions, widely used LLMs remain vulnerable to jailbreak attacks where attackers generate instructions with distributions differing from safety alignment corpora. New attacks expose LLMs' inability to recognize unseen malicious instructions, highlighting a critical distributional mismatch between training data and real-world attacks that forces developers into reactive patching cycles. To tackle this challenge, we propose IMAGINE, a synthesis framework that leverages embedding space distribution analysi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.20038","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-27T16:44:03Z","cross_cats_sorted":[],"title_canon_sha256":"6c41f05a233df787d366c318c0640770637e1e622bcc20492cd4453738bb8b4d","abstract_canon_sha256":"5eed72bc66eb580d9e20a5082d037155cc7b129112a5fa74875c3d84c4f7e8f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:40.134492Z","signature_b64":"74vd8nYJi3RmdqtV8u1aIJVH2u+xS9pTt03ftaQ1tD29gbQvHOK3tBdXBLUZBpJGNBoUNTGVHENOsItRwo9jDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4535c2606896f9768e0c3aeae8e3bee4e9bc61b8b07c389a2a4e6716e3d18f9b","last_reissued_at":"2026-07-05T12:04:40.133988Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:40.133988Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Forewarned is Forearmed: Pre-Synthesizing Jailbreak-like Instructions to Enhance LLM Safety Guardrail to Potential Attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Danding Wang, Guang Yang, Juan Cao, Qiang Sheng, Sheng Liu, Yang Li","submitted_at":"2025-08-27T16:44:03Z","abstract_excerpt":"Despite advances in improving large language model (LLM) to refuse to answer malicious instructions, widely used LLMs remain vulnerable to jailbreak attacks where attackers generate instructions with distributions differing from safety alignment corpora. New attacks expose LLMs' inability to recognize unseen malicious instructions, highlighting a critical distributional mismatch between training data and real-world attacks that forces developers into reactive patching cycles. To tackle this challenge, we propose IMAGINE, a synthesis framework that leverages embedding space distribution analysi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.20038","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.20038/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.20038","created_at":"2026-07-05T12:04:40.134037+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.20038v3","created_at":"2026-07-05T12:04:40.134037+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.20038","created_at":"2026-07-05T12:04:40.134037+00:00"},{"alias_kind":"pith_short_12","alias_value":"IU24EYDIS34X","created_at":"2026-07-05T12:04:40.134037+00:00"},{"alias_kind":"pith_short_16","alias_value":"IU24EYDIS34XNDQM","created_at":"2026-07-05T12:04:40.134037+00:00"},{"alias_kind":"pith_short_8","alias_value":"IU24EYDI","created_at":"2026-07-05T12:04:40.134037+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T","json":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T.json","graph_json":"https://pith.science/api/pith-number/IU24EYDIS34XNDQMHLVORY564T/graph.json","events_json":"https://pith.science/api/pith-number/IU24EYDIS34XNDQMHLVORY564T/events.json","paper":"https://pith.science/paper/IU24EYDI"},"agent_actions":{"view_html":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T","download_json":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T.json","view_paper":"https://pith.science/paper/IU24EYDI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.20038&json=true","fetch_graph":"https://pith.science/api/pith-number/IU24EYDIS34XNDQMHLVORY564T/graph.json","fetch_events":"https://pith.science/api/pith-number/IU24EYDIS34XNDQMHLVORY564T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T/action/storage_attestation","attest_author":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T/action/author_attestation","sign_citation":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T/action/citation_signature","submit_replication":"https://pith.science/pith/IU24EYDIS34XNDQMHLVORY564T/action/replication_record"}},"created_at":"2026-07-05T12:04:40.134037+00:00","updated_at":"2026-07-05T12:04:40.134037+00:00"}