{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LRFMLOLCSJSAT4JJEQELAWXSRG","short_pith_number":"pith:LRFMLOLC","schema_version":"1.0","canonical_sha256":"5c4ac5b962926409f1292408b05af289a82cedb0cf7115e038f0a6d3099e771d","source":{"kind":"arxiv","id":"2411.11114","version":2},"attestation_state":"computed","paper":{"title":"JailbreakLens: Interpreting Jailbreak Mechanism in the Lens of Representation and Circuit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Huiyu Xu, Qinglong Wang, Rui Zheng, Wenhui Zhang, Zeqing He, Zhibo Wang, Zhixuan Chu","submitted_at":"2024-11-17T16:08:34Z","abstract_excerpt":"Despite the outstanding performance of Large language Models (LLMs) in diverse tasks, they are vulnerable to jailbreak attacks, wherein adversarial prompts are crafted to bypass their security mechanisms and elicit unexpected responses. Although jailbreak attacks are prevalent, the understanding of their underlying mechanisms remains limited. Recent studies have explained typical jailbreaking behavior (e.g., the degree to which the model refuses to respond) of LLMs by analyzing representation shifts in their latent space caused by jailbreak prompts or identifying key neurons that contribute to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11114","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-11-17T16:08:34Z","cross_cats_sorted":[],"title_canon_sha256":"e42dcb139acd3ad0a1d12ee1d47a1432123b95bb604dfaf2e59e73d744137e90","abstract_canon_sha256":"d6f6b677d5cb0b612577cd4236fe9e40f27539f96bfd5b7b03e2293210b0fe94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:13.361123Z","signature_b64":"Ff3DhCb24x6CnVNLzjSXQg83vB1lduwkeE/KQ+fSLA3ehrkuk7VNkyh4J1b9lEw5YM6UdRj1TEB8FXiYx8mIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c4ac5b962926409f1292408b05af289a82cedb0cf7115e038f0a6d3099e771d","last_reissued_at":"2026-07-05T10:53:13.360614Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:13.360614Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JailbreakLens: Interpreting Jailbreak Mechanism in the Lens of Representation and Circuit","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Huiyu Xu, Qinglong Wang, Rui Zheng, Wenhui Zhang, Zeqing He, Zhibo Wang, Zhixuan Chu","submitted_at":"2024-11-17T16:08:34Z","abstract_excerpt":"Despite the outstanding performance of Large language Models (LLMs) in diverse tasks, they are vulnerable to jailbreak attacks, wherein adversarial prompts are crafted to bypass their security mechanisms and elicit unexpected responses. Although jailbreak attacks are prevalent, the understanding of their underlying mechanisms remains limited. Recent studies have explained typical jailbreaking behavior (e.g., the degree to which the model refuses to respond) of LLMs by analyzing representation shifts in their latent space caused by jailbreak prompts or identifying key neurons that contribute to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11114","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11114","created_at":"2026-07-05T10:53:13.360677+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11114v2","created_at":"2026-07-05T10:53:13.360677+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11114","created_at":"2026-07-05T10:53:13.360677+00:00"},{"alias_kind":"pith_short_12","alias_value":"LRFMLOLCSJSA","created_at":"2026-07-05T10:53:13.360677+00:00"},{"alias_kind":"pith_short_16","alias_value":"LRFMLOLCSJSAT4JJ","created_at":"2026-07-05T10:53:13.360677+00:00"},{"alias_kind":"pith_short_8","alias_value":"LRFMLOLC","created_at":"2026-07-05T10:53:13.360677+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18104","citing_title":"Safety Geometry Collapse in Multimodal LLMs and Adaptive Drift Correction","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08878","citing_title":"Why Do Aligned LLMs Remain Jailbreakable: Refusal-Escape Directions, Operator-Level Sources, and Safety-Utility Trade-off","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11663","citing_title":"Why Do Large Language Models Generate Harmful Content?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06247","citing_title":"SALLIE: Safeguarding Against Latent Language & Image Exploits","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG","json":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG.json","graph_json":"https://pith.science/api/pith-number/LRFMLOLCSJSAT4JJEQELAWXSRG/graph.json","events_json":"https://pith.science/api/pith-number/LRFMLOLCSJSAT4JJEQELAWXSRG/events.json","paper":"https://pith.science/paper/LRFMLOLC"},"agent_actions":{"view_html":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG","download_json":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG.json","view_paper":"https://pith.science/paper/LRFMLOLC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11114&json=true","fetch_graph":"https://pith.science/api/pith-number/LRFMLOLCSJSAT4JJEQELAWXSRG/graph.json","fetch_events":"https://pith.science/api/pith-number/LRFMLOLCSJSAT4JJEQELAWXSRG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG/action/storage_attestation","attest_author":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG/action/author_attestation","sign_citation":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG/action/citation_signature","submit_replication":"https://pith.science/pith/LRFMLOLCSJSAT4JJEQELAWXSRG/action/replication_record"}},"created_at":"2026-07-05T10:53:13.360677+00:00","updated_at":"2026-07-05T10:53:13.360677+00:00"}