{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:APOZRYMUH7JSEW32OKK6RQ5AN4","short_pith_number":"pith:APOZRYMU","schema_version":"1.0","canonical_sha256":"03dd98e1943fd3225b7a7295e8c3a06f2d7d16860084b2915713ecf05b4f9d55","source":{"kind":"arxiv","id":"2401.06824","version":5},"attestation_state":"computed","paper":{"title":"Revisiting Jailbreaking for Large Language Models: A Representation Engineering Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Changze Lv, Muling Wu, Shihan Dou, Tianlong Li, Wenhao Liu, Xiaohua Wang, Xiaoqing Zheng, Xuanjing Huang, Zhenghua Wang","submitted_at":"2024-01-12T00:50:04Z","abstract_excerpt":"The recent surge in jailbreaking attacks has revealed significant vulnerabilities in Large Language Models (LLMs) when exposed to malicious inputs. While various defense strategies have been proposed to mitigate these threats, there has been limited research into the underlying mechanisms that make LLMs vulnerable to such attacks. In this study, we suggest that the self-safeguarding capability of LLMs is linked to specific activity patterns within their representation space. Although these patterns have little impact on the semantic content of the generated text, they play a crucial role in sh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06824","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-12T00:50:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fa220830f33e48c64ae27de8ba63d56b7d74dd93755ce00e0def449027711639","abstract_canon_sha256":"327b2d300ca05604b6731a2dd29afa538406a6a041a0d19ff0c386eb2f437b8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:45.853413Z","signature_b64":"V8pYi5uq/GbXAjwHObhB6BQQUDBW1m0EoWFiONF9aWct77MmXBr6dyAL/LOeLMNECgxnPb3p+IITB/S7IcrlAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03dd98e1943fd3225b7a7295e8c3a06f2d7d16860084b2915713ecf05b4f9d55","last_reissued_at":"2026-07-05T10:17:45.852883Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:45.852883Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Jailbreaking for Large Language Models: A Representation Engineering Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Changze Lv, Muling Wu, Shihan Dou, Tianlong Li, Wenhao Liu, Xiaohua Wang, Xiaoqing Zheng, Xuanjing Huang, Zhenghua Wang","submitted_at":"2024-01-12T00:50:04Z","abstract_excerpt":"The recent surge in jailbreaking attacks has revealed significant vulnerabilities in Large Language Models (LLMs) when exposed to malicious inputs. While various defense strategies have been proposed to mitigate these threats, there has been limited research into the underlying mechanisms that make LLMs vulnerable to such attacks. In this study, we suggest that the self-safeguarding capability of LLMs is linked to specific activity patterns within their representation space. Although these patterns have little impact on the semantic content of the generated text, they play a crucial role in sh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06824","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06824/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06824","created_at":"2026-07-05T10:17:45.852944+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06824v5","created_at":"2026-07-05T10:17:45.852944+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06824","created_at":"2026-07-05T10:17:45.852944+00:00"},{"alias_kind":"pith_short_12","alias_value":"APOZRYMUH7JS","created_at":"2026-07-05T10:17:45.852944+00:00"},{"alias_kind":"pith_short_16","alias_value":"APOZRYMUH7JSEW32","created_at":"2026-07-05T10:17:45.852944+00:00"},{"alias_kind":"pith_short_8","alias_value":"APOZRYMU","created_at":"2026-07-05T10:17:45.852944+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18104","citing_title":"Safety Geometry Collapse in Multimodal LLMs and Adaptive Drift Correction","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4","json":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4.json","graph_json":"https://pith.science/api/pith-number/APOZRYMUH7JSEW32OKK6RQ5AN4/graph.json","events_json":"https://pith.science/api/pith-number/APOZRYMUH7JSEW32OKK6RQ5AN4/events.json","paper":"https://pith.science/paper/APOZRYMU"},"agent_actions":{"view_html":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4","download_json":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4.json","view_paper":"https://pith.science/paper/APOZRYMU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06824&json=true","fetch_graph":"https://pith.science/api/pith-number/APOZRYMUH7JSEW32OKK6RQ5AN4/graph.json","fetch_events":"https://pith.science/api/pith-number/APOZRYMUH7JSEW32OKK6RQ5AN4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4/action/storage_attestation","attest_author":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4/action/author_attestation","sign_citation":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4/action/citation_signature","submit_replication":"https://pith.science/pith/APOZRYMUH7JSEW32OKK6RQ5AN4/action/replication_record"}},"created_at":"2026-07-05T10:17:45.852944+00:00","updated_at":"2026-07-05T10:17:45.852944+00:00"}