{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PB4Y5OQLVNKV7RFRC7GJTEBX5M","short_pith_number":"pith:PB4Y5OQL","schema_version":"1.0","canonical_sha256":"78798eba0bab555fc4b117cc999037eb171917b49a564752f23fd71c36f063d2","source":{"kind":"arxiv","id":"2312.07130","version":4},"attestation_state":"computed","paper":{"title":"Harnessing LLM to Attack LLM-Guarded Text-to-Image Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Huangxun Chen, Yimo Deng","submitted_at":"2023-12-12T10:04:43Z","abstract_excerpt":"To prevent Text-to-Image (T2I) models from generating unethical images, people deploy safety filters to block inappropriate drawing prompts. Previous works have employed token replacement to search adversarial prompts that attempt to bypass these filters, but they have become ineffective as nonsensical tokens fail semantic logic checks. In this paper, we approach adversarial prompts from a different perspective. We demonstrate that rephrasing a drawing intent into multiple benign descriptions of individual visual components can obtain an effective adversarial prompt. We propose a LLM-piloted m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.07130","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-12-12T10:04:43Z","cross_cats_sorted":[],"title_canon_sha256":"3901b16862f823cc533cab0a50638de985b3862b38d3dc6fe8ac325ff350758e","abstract_canon_sha256":"d3364f90e17f40724b4ac7672995d4c2968f4eaab979de89709e6223f73b59ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:40:15.731117Z","signature_b64":"vQ17F4Mmy2JV9T9E5LpyDf0Yd7Ez1u/UIn+Ohg6O8YLeDFM+cGtTJ28kjDTwr4gXKerfpG4GrRXDwrNAabdsBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78798eba0bab555fc4b117cc999037eb171917b49a564752f23fd71c36f063d2","last_reissued_at":"2026-07-05T09:40:15.730593Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:40:15.730593Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Harnessing LLM to Attack LLM-Guarded Text-to-Image Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Huangxun Chen, Yimo Deng","submitted_at":"2023-12-12T10:04:43Z","abstract_excerpt":"To prevent Text-to-Image (T2I) models from generating unethical images, people deploy safety filters to block inappropriate drawing prompts. Previous works have employed token replacement to search adversarial prompts that attempt to bypass these filters, but they have become ineffective as nonsensical tokens fail semantic logic checks. In this paper, we approach adversarial prompts from a different perspective. We demonstrate that rephrasing a drawing intent into multiple benign descriptions of individual visual components can obtain an effective adversarial prompt. We propose a LLM-piloted m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.07130","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.07130/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.07130","created_at":"2026-07-05T09:40:15.730657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.07130v4","created_at":"2026-07-05T09:40:15.730657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.07130","created_at":"2026-07-05T09:40:15.730657+00:00"},{"alias_kind":"pith_short_12","alias_value":"PB4Y5OQLVNKV","created_at":"2026-07-05T09:40:15.730657+00:00"},{"alias_kind":"pith_short_16","alias_value":"PB4Y5OQLVNKV7RFR","created_at":"2026-07-05T09:40:15.730657+00:00"},{"alias_kind":"pith_short_8","alias_value":"PB4Y5OQL","created_at":"2026-07-05T09:40:15.730657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09151","citing_title":"Customization under Fire: Plugin Poisoning in Text-to-Image Ecosystem","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01837","citing_title":"Benign Inputs, Harmful Outputs: Cross-Modal Jailbreaking via Distributed Semantic Recomposition","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26332","citing_title":"Erased but Exploitable: Black-box Embedding-Aware Prompting Against Unlearned Text-to-Image Diffusion Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01481","citing_title":"SafeGen-Bench: Benchmarking Safety in Image-Conditioned Text-to-Video Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01113","citing_title":"Disciplined Diffusion: Text-to-Image Diffusion Model against NSFW Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01761","citing_title":"TrajShield: Trajectory-Level Safety Mediation for Defending Text-to-Video Models Against Jailbreak Attacks","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M","json":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M.json","graph_json":"https://pith.science/api/pith-number/PB4Y5OQLVNKV7RFRC7GJTEBX5M/graph.json","events_json":"https://pith.science/api/pith-number/PB4Y5OQLVNKV7RFRC7GJTEBX5M/events.json","paper":"https://pith.science/paper/PB4Y5OQL"},"agent_actions":{"view_html":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M","download_json":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M.json","view_paper":"https://pith.science/paper/PB4Y5OQL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.07130&json=true","fetch_graph":"https://pith.science/api/pith-number/PB4Y5OQLVNKV7RFRC7GJTEBX5M/graph.json","fetch_events":"https://pith.science/api/pith-number/PB4Y5OQLVNKV7RFRC7GJTEBX5M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M/action/storage_attestation","attest_author":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M/action/author_attestation","sign_citation":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M/action/citation_signature","submit_replication":"https://pith.science/pith/PB4Y5OQLVNKV7RFRC7GJTEBX5M/action/replication_record"}},"created_at":"2026-07-05T09:40:15.730657+00:00","updated_at":"2026-07-05T09:40:15.730657+00:00"}