{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DNLG7ZJDAP5HUMXO6IL63Q24ML","short_pith_number":"pith:DNLG7ZJD","schema_version":"1.0","canonical_sha256":"1b566fe52303fa7a32eef217edc35c62ffd0bd9011ee1c41ac54a6f244da5621","source":{"kind":"arxiv","id":"2503.20823","version":1},"attestation_state":"computed","paper":{"title":"Playing the Fool: Jailbreaking LLMs and Multimodal LLMs with Out-of-Distribution Strategy","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Eunho Yang, Jaeryong Hwang, Joonhyun Jeong, Seyun Bae, Yeonsung Jung","submitted_at":"2025-03-26T01:25:24Z","abstract_excerpt":"Despite the remarkable versatility of Large Language Models (LLMs) and Multimodal LLMs (MLLMs) to generalize across both language and vision tasks, LLMs and MLLMs have shown vulnerability to jailbreaking, generating textual outputs that undermine safety, ethical, and bias standards when exposed to harmful or sensitive inputs. With the recent advancement of safety alignment via preference-tuning from human feedback, LLMs and MLLMs have been equipped with safety guardrails to yield safe, ethical, and fair responses with regard to harmful inputs. However, despite the significance of safety alignm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20823","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CR","submitted_at":"2025-03-26T01:25:24Z","cross_cats_sorted":[],"title_canon_sha256":"7448890cee98f218c331e4eefd4b1f912df44ac569f6bc81ce0fd792496b1e4e","abstract_canon_sha256":"748dd9a441b96b072fc92c5bec27f48f15b84bad8d7b9e9bcf6e3a6d51c7524a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:48.754613Z","signature_b64":"u1G6Am65obzZ9uFduWWHArOkdHDHRJ8svEB/s2hGyG8FDl5mwVLAjcAeYJ7LfjeGxgn93nzUDMnlwSsOdPZwDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b566fe52303fa7a32eef217edc35c62ffd0bd9011ee1c41ac54a6f244da5621","last_reissued_at":"2026-07-05T10:39:48.754129Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:48.754129Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Playing the Fool: Jailbreaking LLMs and Multimodal LLMs with Out-of-Distribution Strategy","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Eunho Yang, Jaeryong Hwang, Joonhyun Jeong, Seyun Bae, Yeonsung Jung","submitted_at":"2025-03-26T01:25:24Z","abstract_excerpt":"Despite the remarkable versatility of Large Language Models (LLMs) and Multimodal LLMs (MLLMs) to generalize across both language and vision tasks, LLMs and MLLMs have shown vulnerability to jailbreaking, generating textual outputs that undermine safety, ethical, and bias standards when exposed to harmful or sensitive inputs. With the recent advancement of safety alignment via preference-tuning from human feedback, LLMs and MLLMs have been equipped with safety guardrails to yield safe, ethical, and fair responses with regard to harmful inputs. However, despite the significance of safety alignm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20823","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20823/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20823","created_at":"2026-07-05T10:39:48.754183+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20823v1","created_at":"2026-07-05T10:39:48.754183+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20823","created_at":"2026-07-05T10:39:48.754183+00:00"},{"alias_kind":"pith_short_12","alias_value":"DNLG7ZJDAP5H","created_at":"2026-07-05T10:39:48.754183+00:00"},{"alias_kind":"pith_short_16","alias_value":"DNLG7ZJDAP5HUMXO","created_at":"2026-07-05T10:39:48.754183+00:00"},{"alias_kind":"pith_short_8","alias_value":"DNLG7ZJD","created_at":"2026-07-05T10:39:48.754183+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML","json":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML.json","graph_json":"https://pith.science/api/pith-number/DNLG7ZJDAP5HUMXO6IL63Q24ML/graph.json","events_json":"https://pith.science/api/pith-number/DNLG7ZJDAP5HUMXO6IL63Q24ML/events.json","paper":"https://pith.science/paper/DNLG7ZJD"},"agent_actions":{"view_html":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML","download_json":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML.json","view_paper":"https://pith.science/paper/DNLG7ZJD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20823&json=true","fetch_graph":"https://pith.science/api/pith-number/DNLG7ZJDAP5HUMXO6IL63Q24ML/graph.json","fetch_events":"https://pith.science/api/pith-number/DNLG7ZJDAP5HUMXO6IL63Q24ML/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML/action/storage_attestation","attest_author":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML/action/author_attestation","sign_citation":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML/action/citation_signature","submit_replication":"https://pith.science/pith/DNLG7ZJDAP5HUMXO6IL63Q24ML/action/replication_record"}},"created_at":"2026-07-05T10:39:48.754183+00:00","updated_at":"2026-07-05T10:39:48.754183+00:00"}