{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JYM7UZ4ACXTSYU4YMU6QUC3L2R","short_pith_number":"pith:JYM7UZ4A","schema_version":"1.0","canonical_sha256":"4e19fa678015e72c5398653d0a0b6bd45e5881e6834685e44d309aa38cdfe77f","source":{"kind":"arxiv","id":"2406.05644","version":2},"attestation_state":"computed","paper":{"title":"How Alignment and Jailbreak Work: Explain LLM Safety through Intermediate Hidden States","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.CY"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Haiyang Yu, Rongwu Xu, Xinghua Zhang, Yongbin Li, Zhenhong Zhou","submitted_at":"2024-06-09T05:04:37Z","abstract_excerpt":"Large language models (LLMs) rely on safety alignment to avoid responding to malicious user inputs. Unfortunately, jailbreak can circumvent safety guardrails, resulting in LLMs generating harmful content and raising concerns about LLM safety. Due to language models with intensive parameters often regarded as black boxes, the mechanisms of alignment and jailbreak are challenging to elucidate. In this paper, we employ weak classifiers to explain LLM safety through the intermediate hidden states. We first confirm that LLMs learn ethical concepts during pre-training rather than alignment and can i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05644","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-09T05:04:37Z","cross_cats_sorted":["cs.AI","cs.CR","cs.CY"],"title_canon_sha256":"0c9c82d1d8aac8f68d7da6ecf925ca6eb533b08fee4b8979043f316ab867fab2","abstract_canon_sha256":"7920e54f5969de4dbbe044f81efe7bc1e21d97c91b79be996b8b98785eceb5a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:16.378905Z","signature_b64":"VV2Z6c9gb+89nSPQMKXowbrFOgNuKrURMA6VFtBA+IF0w1WH93Dze6eVYTBCjC9CaviZ7vjcNKn8DlDA7DbrAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e19fa678015e72c5398653d0a0b6bd45e5881e6834685e44d309aa38cdfe77f","last_reissued_at":"2026-07-05T08:31:16.378420Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:16.378420Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Alignment and Jailbreak Work: Explain LLM Safety through Intermediate Hidden States","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.CY"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Haiyang Yu, Rongwu Xu, Xinghua Zhang, Yongbin Li, Zhenhong Zhou","submitted_at":"2024-06-09T05:04:37Z","abstract_excerpt":"Large language models (LLMs) rely on safety alignment to avoid responding to malicious user inputs. Unfortunately, jailbreak can circumvent safety guardrails, resulting in LLMs generating harmful content and raising concerns about LLM safety. Due to language models with intensive parameters often regarded as black boxes, the mechanisms of alignment and jailbreak are challenging to elucidate. In this paper, we employ weak classifiers to explain LLM safety through the intermediate hidden states. We first confirm that LLMs learn ethical concepts during pre-training rather than alignment and can i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05644","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05644","created_at":"2026-07-05T08:31:16.378476+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05644v2","created_at":"2026-07-05T08:31:16.378476+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05644","created_at":"2026-07-05T08:31:16.378476+00:00"},{"alias_kind":"pith_short_12","alias_value":"JYM7UZ4ACXTS","created_at":"2026-07-05T08:31:16.378476+00:00"},{"alias_kind":"pith_short_16","alias_value":"JYM7UZ4ACXTSYU4Y","created_at":"2026-07-05T08:31:16.378476+00:00"},{"alias_kind":"pith_short_8","alias_value":"JYM7UZ4A","created_at":"2026-07-05T08:31:16.378476+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08423","citing_title":"OmniFood-Bench: Evaluating VLMs for Nutrient Reasoning and Personalized Health Advice","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19755","citing_title":"SafeSpec: Fast and Safe LLM via Dynamic Reflective Sampling","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12726","citing_title":"Before the Last Token: Diagnosing Final-Token Safety Probe Failures","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11663","citing_title":"Why Do Large Language Models Generate Harmful Content?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02914","citing_title":"When Safety Geometry Collapses: Fine-Tuning Vulnerabilities in Agentic Guard Models","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R","json":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R.json","graph_json":"https://pith.science/api/pith-number/JYM7UZ4ACXTSYU4YMU6QUC3L2R/graph.json","events_json":"https://pith.science/api/pith-number/JYM7UZ4ACXTSYU4YMU6QUC3L2R/events.json","paper":"https://pith.science/paper/JYM7UZ4A"},"agent_actions":{"view_html":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R","download_json":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R.json","view_paper":"https://pith.science/paper/JYM7UZ4A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05644&json=true","fetch_graph":"https://pith.science/api/pith-number/JYM7UZ4ACXTSYU4YMU6QUC3L2R/graph.json","fetch_events":"https://pith.science/api/pith-number/JYM7UZ4ACXTSYU4YMU6QUC3L2R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R/action/storage_attestation","attest_author":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R/action/author_attestation","sign_citation":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R/action/citation_signature","submit_replication":"https://pith.science/pith/JYM7UZ4ACXTSYU4YMU6QUC3L2R/action/replication_record"}},"created_at":"2026-07-05T08:31:16.378476+00:00","updated_at":"2026-07-05T08:31:16.378476+00:00"}