{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LRTMBMDMIEP7AFPPMDNYGEWX5S","short_pith_number":"pith:LRTMBMDM","schema_version":"1.0","canonical_sha256":"5c66c0b06c411ff015ef60db8312d7ecbe372f51bd01c441f00826902aaeb512","source":{"kind":"arxiv","id":"2402.08679","version":2},"attestation_state":"computed","paper":{"title":"COLD-Attack: Jailbreaking LLMs with Stealthiness and Controllability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bin Hu, Fangxu Yu, Huan Zhang, Lianhui Qin, Xingang Guo","submitted_at":"2024-02-13T18:58:48Z","abstract_excerpt":"Jailbreaks on large language models (LLMs) have recently received increasing attention. For a comprehensive assessment of LLM safety, it is essential to consider jailbreaks with diverse attributes, such as contextual coherence and sentiment/stylistic variations, and hence it is beneficial to study controllable jailbreaking, i.e. how to enforce control on LLM attacks. In this paper, we formally formulate the controllable attack generation problem, and build a novel connection between this problem and controllable text generation, a well-explored topic of natural language processing. Based on th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08679","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-13T18:58:48Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"0d18948850546066965693c19689741cac5af9242b9dd69296f72797f7932182","abstract_canon_sha256":"c287f41d86e25e82860946a778ce050e3c6d6f431a3413d7a1a4a7564190efe4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:38.092691Z","signature_b64":"yeUdVMj9EVJ+KsvHDbRywXTq8zUKKzBu8P8Z+1sG9NCOTJHbuX7L/APt2VAJvbLTskOr49U/M4qT2FrL0EYOAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c66c0b06c411ff015ef60db8312d7ecbe372f51bd01c441f00826902aaeb512","last_reissued_at":"2026-07-05T08:28:38.092265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:38.092265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"COLD-Attack: Jailbreaking LLMs with Stealthiness and Controllability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bin Hu, Fangxu Yu, Huan Zhang, Lianhui Qin, Xingang Guo","submitted_at":"2024-02-13T18:58:48Z","abstract_excerpt":"Jailbreaks on large language models (LLMs) have recently received increasing attention. For a comprehensive assessment of LLM safety, it is essential to consider jailbreaks with diverse attributes, such as contextual coherence and sentiment/stylistic variations, and hence it is beneficial to study controllable jailbreaking, i.e. how to enforce control on LLM attacks. In this paper, we formally formulate the controllable attack generation problem, and build a novel connection between this problem and controllable text generation, a well-explored topic of natural language processing. Based on th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08679","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08679/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08679","created_at":"2026-07-05T08:28:38.092329+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08679v2","created_at":"2026-07-05T08:28:38.092329+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08679","created_at":"2026-07-05T08:28:38.092329+00:00"},{"alias_kind":"pith_short_12","alias_value":"LRTMBMDMIEP7","created_at":"2026-07-05T08:28:38.092329+00:00"},{"alias_kind":"pith_short_16","alias_value":"LRTMBMDMIEP7AFPP","created_at":"2026-07-05T08:28:38.092329+00:00"},{"alias_kind":"pith_short_8","alias_value":"LRTMBMDM","created_at":"2026-07-05T08:28:38.092329+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2405.13068","citing_title":"Uncovering Logit Suppression Vulnerabilities in LLM Safety Alignment","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.00446","citing_title":"Exposing the Ghost in the Transformer: Abnormal Detection for Large Language Models via Hidden State Forensics","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21674","citing_title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21362","citing_title":"LASH: Adaptive Semantic Hybridization for Black-Box Jailbreaking of Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02356","citing_title":"ASTRA: An Automated Framework for Strategy Discovery, Retrieval, and Evolution for Jailbreaking LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12232","citing_title":"TEMPLATEFUZZ: Fine-Grained Chat Template Fuzzing for Jailbreaking and Red Teaming LLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08881","citing_title":"Targeted Interpretable Safety Neuron Enhancement for Multilingual Vision-Language Large Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18976","citing_title":"STAR-Teaming: A Strategy-Response Multiplex Network Approach to Automated LLM Red Teaming","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S","json":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S.json","graph_json":"https://pith.science/api/pith-number/LRTMBMDMIEP7AFPPMDNYGEWX5S/graph.json","events_json":"https://pith.science/api/pith-number/LRTMBMDMIEP7AFPPMDNYGEWX5S/events.json","paper":"https://pith.science/paper/LRTMBMDM"},"agent_actions":{"view_html":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S","download_json":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S.json","view_paper":"https://pith.science/paper/LRTMBMDM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08679&json=true","fetch_graph":"https://pith.science/api/pith-number/LRTMBMDMIEP7AFPPMDNYGEWX5S/graph.json","fetch_events":"https://pith.science/api/pith-number/LRTMBMDMIEP7AFPPMDNYGEWX5S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S/action/storage_attestation","attest_author":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S/action/author_attestation","sign_citation":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S/action/citation_signature","submit_replication":"https://pith.science/pith/LRTMBMDMIEP7AFPPMDNYGEWX5S/action/replication_record"}},"created_at":"2026-07-05T08:28:38.092329+00:00","updated_at":"2026-07-05T08:28:38.092329+00:00"}