{"as_of":"2026-08-08T06:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a374a7fce4ee43b7baffe790c325dd642738047a915128349d2c82ae417da303","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T23:08:02.302336Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2409.18169","last_updated":"2026-04-23T18:48:49Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:55:22Z","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","version":6},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-23T20:58:16.237327Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2409.18169"},"observation_digest":"sha256:9e65827169f789f4f71a54e2707ab2f16877fcc9045eb116d6c0665e8ed1c15e","observation_id":"c9d6c6d8-a600-496a-9fc6-c334deb18f87","resolution":{"observed_at":"2026-05-23T20:58:26.349920Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-07T23:08:02.302336Z","title":"Safety layers in aligned large language models: The key to llm security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09674","last_updated":"2025-05-27T08:40:42Z","snapshot_observed_at":"2026-08-07T22:55:15.992320Z","submitted_at":"2025-02-13T06:39:22Z","title":"The Hidden Dimensions of LLM Alignment: A Multi-Dimensional Analysis of Orthogonal Safety Directions","version":4},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T23:08:02.302336Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2502.09674"},"observation_digest":"sha256:127e3c50239a5d2964b951b0ea6ecd8c3876bc1407f3d9f0ce6ef4065d406b34","observation_id":"9a175d9f-deb9-4e8c-8ceb-07f6bee309d0","resolution":{"observed_at":"2026-08-07T23:08:02.302336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-07T14:15:19.536862Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19670","last_updated":"2025-05-26T08:25:25Z","snapshot_observed_at":"2026-08-07T14:06:45.672196Z","submitted_at":"2025-05-26T08:25:25Z","title":"Reshaping Representation Space to Balance the Safety and Over-rejection in Large Audio Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T14:15:19.536862Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2505.19670"},"observation_digest":"sha256:986afe33b98ff5c14a5587191e457694441f800c3b100dcfe09b0ee29818651f","observation_id":"7dd8fe59-9419-43ff-9ff5-03d46611f82f","resolution":{"observed_at":"2026-08-07T14:15:19.536862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-06T19:57:43.958573Z","title":"Safety layers in aligned large language models: The key to llm security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04250","last_updated":"2025-07-06T05:47:04Z","snapshot_observed_at":"2026-08-07T18:43:56.210003Z","submitted_at":"2025-07-06T05:47:04Z","title":"Just Enough Shifts: Mitigating Over-Refusal in Aligned Language Models with Targeted Representation Fine-Tuning","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T19:57:43.958573Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2507.04250"},"observation_digest":"sha256:8d9dc1b95d04ffa71e1b7b80547fdd6ea8b2d2a15dcdb03f4a554a3b1d920216","observation_id":"7cc02672-ba13-4cb4-8174-ab22e8cf3187","resolution":{"observed_at":"2026-08-06T19:57:43.958573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-06T15:18:11.656139Z","title":"Safety layers of aligned large language models: The key to llm security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16372","last_updated":"2025-07-22T09:15:11Z","snapshot_observed_at":"2026-08-06T15:09:00.559058Z","submitted_at":"2025-07-22T09:15:11Z","title":"Depth Gives a False Sense of Privacy: LLM Internal States Inversion","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:18:11.656139Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2507.16372"},"observation_digest":"sha256:0e9a5c3f488f5090c669189f5b8ff294ac6e606564dee22797b0661d348b2a90","observation_id":"24787934-f6cc-4630-b1bd-907232ad4328","resolution":{"observed_at":"2026-08-06T15:18:11.656139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T12:39:34.079317Z","title":"Safety layers in aligned large language models: The key to llm security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01414","last_updated":"2025-09-01T12:11:42Z","snapshot_observed_at":"2026-08-05T12:39:32.700175Z","submitted_at":"2025-09-01T12:11:42Z","title":"AttenTrack: Mobile User Attention Awareness Based on Context and External Distractions","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T12:39:34.079317Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2509.01414"},"observation_digest":"sha256:bc50ec258d27a3aa16374ff455ac363eaa7f84998f655f6f5866a12ce2ce1fc9","observation_id":"584c895b-3c08-4de2-92b6-025324c04dff","resolution":{"observed_at":"2026-08-05T12:39:34.079317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T06:00:51.098948Z","title":"Safety layers of aligned large language models: The key to llm security,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.04714","last_updated":"2025-09-05T00:02:17Z","snapshot_observed_at":"2026-08-08T05:15:31.834461Z","submitted_at":"2025-09-05T00:02:17Z","title":"ThumbnailTruth: A Multi-Modal LLM Approach for Detecting Misleading YouTube Thumbnails Across Diverse Cultural Settings","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T06:00:51.098948Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2509.04714"},"observation_digest":"sha256:8f8f166f34137c11173a0f5c9a43b6ca2ec6931891a9ff152c4769e404e58147","observation_id":"69d67a9a-514b-438a-90db-0a0fd96c33e1","resolution":{"observed_at":"2026-08-05T06:00:51.098948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-04T23:10:21.170218Z","title":"Safety layers in aligned large language models: The key to llm security,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.06795","last_updated":"2025-09-08T15:24:33Z","snapshot_observed_at":"2026-08-06T11:18:33.345942Z","submitted_at":"2025-09-08T15:24:33Z","title":"Anchoring Refusal Direction: Mitigating Safety Risks in Tuning via Projection Constraint","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T23:10:21.170218Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2509.06795"},"observation_digest":"sha256:ec15b84467335f4ce96b2b66c824674252e646ff7d59625c78a1a484ae867892","observation_id":"e084c66c-68b0-4f02-874b-7e9a41c635bd","resolution":{"observed_at":"2026-08-04T23:10:21.170218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2511.06516","last_updated":"2026-06-28T18:20:13Z","snapshot_observed_at":"2026-08-03T23:21:36.698338Z","submitted_at":"2025-11-09T19:58:24Z","title":"You Had One Job: Per-Task Quantization Using LLMs' Hidden Representations","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-21T18:46:04.926179Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2511.06516"},"observation_digest":"sha256:b1d18abeed86a9327989a95d3c2a9f77a1f44520d6a2abe39bdc0f41e2a41336","observation_id":"3b9e438f-ad96-431b-a1f4-ed0d470b50f6","resolution":{"observed_at":"2026-05-21T18:50:30.239698Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-03T23:21:39.668021Z","title":"Safety layers in aligned large language models: The key to llm security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.06516","last_updated":"2026-06-28T18:20:13Z","snapshot_observed_at":"2026-08-03T23:21:36.698338Z","submitted_at":"2025-11-09T19:58:24Z","title":"You Had One Job: Per-Task Quantization Using LLMs' Hidden Representations","version":4},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-03T23:21:39.668021Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2511.06516"},"observation_digest":"sha256:6481680b906347dfc24ce829a109296d43df802c1fccf5beab1c40c4a6732181","observation_id":"1c1476a3-099a-4a33-9b61-d5606d6db2b1","resolution":{"observed_at":"2026-08-03T23:21:39.668021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2602.15853","last_updated":"2026-04-27T00:25:03Z","snapshot_observed_at":"2026-07-31T20:33:52.305319Z","submitted_at":"2026-01-24T03:58:45Z","title":"A Lightweight Explainable Guardrail for Prompt Safety","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:14.226584Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2602.15853"},"observation_digest":"sha256:0cdf4ec173b11918e0227bc628be4a1ea3ebb935051b7a39aabc1065ff5c3cfb","observation_id":"3481c5ea-0f6b-4fd7-9030-6b18cc4775ad","resolution":{"observed_at":"2026-05-16T11:47:49.345374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2604.06247","last_updated":"2026-04-06T16:29:05Z","snapshot_observed_at":"2026-07-06T22:54:48.916799Z","submitted_at":"2026-04-06T16:29:05Z","title":"SALLIE: Safeguarding Against Latent Language & Image Exploits","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T19:04:46.426969Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2604.06247"},"observation_digest":"sha256:9992897f08aa0da8b9eae5c44f3ca76de4e1c8968b30c032427c2a5208b7c8e2","observation_id":"f77c2ff1-48ca-4d32-affb-ba58a833a3d3","resolution":{"observed_at":"2026-05-10T23:30:51.234243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2604.11663","last_updated":"2026-04-13T16:11:38Z","snapshot_observed_at":"2026-07-06T22:59:59.220594Z","submitted_at":"2026-04-13T16:11:38Z","title":"Why Do Large Language Models Generate Harmful Content?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T15:31:13.545599Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2604.11663"},"observation_digest":"sha256:0e3aa517b88e1f500860a974b917acd14725115e8c8cb85a5d435bf7635743ef","observation_id":"139f1aa5-5823-4014-8e25-866da150b6a3","resolution":{"observed_at":"2026-05-11T10:21:04.499967Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2604.12384","last_updated":"2026-04-14T07:17:55Z","snapshot_observed_at":"2026-07-06T23:00:37.144068Z","submitted_at":"2026-04-14T07:17:55Z","title":"Preventing Safety Drift in Large Language Models via Coupled Weight and Activation Constraints","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-10T16:04:25.851592Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2604.12384"},"observation_digest":"sha256:85ec74cbc7e71e261eba05f8badc2b3abe830bc0293033f8459b864f1a2568d3","observation_id":"1587fbe5-177a-4d03-94cf-e5ceb4ca1ca0","resolution":{"observed_at":"2026-05-11T09:21:01.769674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:b6ea14b8e92b99ffbe0eaae7488d687d1ca799e10684ece9b26b8b453ba74206","observation_id":"d09d643a-6e56-4864-a8ff-6dafdace5ff8","resolution":{"observed_at":"2026-05-10T12:20:23.070753Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2605.10998","last_updated":"2026-05-09T15:52:29Z","snapshot_observed_at":"2026-08-03T01:46:31.309604Z","submitted_at":"2026-05-09T15:52:29Z","title":"Few-Shot Truly Benign DPO Attack for Jailbreaking LLMs","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-13T07:06:46.387088Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2605.10998"},"observation_digest":"sha256:1299113da4c959c93e0a617aa24f087a99a0df39b8a5702039993d4dd3fb72c0","observation_id":"754ebd1a-ce5c-4832-a16a-f49f72bf005a","resolution":{"observed_at":"2026-05-13T07:07:26.976711Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2605.14514","last_updated":"2026-05-14T07:58:47Z","snapshot_observed_at":"2026-07-06T23:25:54.070869Z","submitted_at":"2026-05-14T07:58:47Z","title":"Defenses at Odds: Measuring and Explaining Defense Conflicts in Large Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-15T01:43:22.777232Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2605.14514"},"observation_digest":"sha256:aadb527d1c8c6e5f779abd13db6a068e2a7902aef820ac42a174cdde86bef787","observation_id":"9fb2d29e-5e4b-4ff3-944f-99559e102d70","resolution":{"observed_at":"2026-05-15T01:43:27.337237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2606.07335","last_updated":"2026-06-05T14:49:26Z","snapshot_observed_at":"2026-07-06T23:47:00.425715Z","submitted_at":"2026-06-05T14:49:26Z","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T21:55:48.561400Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2606.07335"},"observation_digest":"sha256:b789e801df90dd3a9fd012b3d36cdfe1c25702a1b64b24cc7c52c64733a2788e","observation_id":"21cc8787-5edb-4d31-ac65-ae4b255935fd","resolution":{"observed_at":"2026-07-02T17:37:14.893132Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2606.15980","last_updated":"2026-06-18T23:56:29Z","snapshot_observed_at":"2026-07-06T23:52:37.922109Z","submitted_at":"2026-06-14T19:07:22Z","title":"Do Activation Monitors Survive Model Updates? Benchmarking, Predicting, and Repairing Activation-Monitor Staleness","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-06-27T03:24:24.714121Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2606.15980"},"observation_digest":"sha256:086aff49148de6b8c7c59e26bca186f41744e5586697e11e9e68da4a8fb53ff7","observation_id":"b5a85a15-ae91-47a9-b963-c642c9be3b3e","resolution":{"observed_at":"2026-06-27T03:30:26.988520Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2606.28153","last_updated":"2026-06-29T15:00:34Z","snapshot_observed_at":"2026-07-07T00:02:18.989389Z","submitted_at":"2026-06-26T14:51:16Z","title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-06-29T03:35:34.594617Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2606.28153"},"observation_digest":"sha256:1f9eae959bf4cc2ba68e89c6c1d0415a4949e9fe3c031fee0554616c3198fe55","observation_id":"986a73b8-f404-4c7f-a866-6e054d114540","resolution":{"observed_at":"2026-07-01T17:35:51.328901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2606.28153","last_updated":"2026-06-29T15:00:34Z","snapshot_observed_at":"2026-07-07T00:02:18.989389Z","submitted_at":"2026-06-26T14:51:16Z","title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-06-30T09:32:59.824110Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2606.28153"},"observation_digest":"sha256:94ced229c9f924ac8c8ccb10282c3c8f80b7b92b3895f91820fbe601af77d5c1","observation_id":"9a408c1a-a20c-44a1-9484-7a1639be7c37","resolution":{"observed_at":"2026-06-30T09:34:34.453497Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-01T17:41:11.909026Z","title":"arXiv preprint arXiv:2408.17003 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17575","last_updated":"2026-07-20T05:42:25Z","snapshot_observed_at":"2026-08-06T23:47:03.742852Z","submitted_at":"2026-07-20T05:42:25Z","title":"A Dual-Hypothesis Reasoning Framework for LLM Guardrails","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-01T17:41:11.909026Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2607.17575"},"observation_digest":"sha256:81d19d19f13b406c2e847a9e824fe764603c0028fdd935b61c006fc413333644","observation_id":"1bc1f87d-385a-4433-91f2-33b429b9c815","resolution":{"observed_at":"2026-08-01T17:41:11.909026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-01T12:04:24.102817Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19683","last_updated":"2026-07-22T02:33:56Z","snapshot_observed_at":"2026-08-07T20:47:58.956911Z","submitted_at":"2026-07-22T02:33:56Z","title":"GhostPrompt: Cross-Image Adversarial Prompt for Vision-Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T12:04:24.102817Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2607.19683"},"observation_digest":"sha256:89e8a0dcc3453c016e28f59b670e5ff05a478d6ff23429d90f18df00669abd16","observation_id":"f418fcab-282e-49ce-a78c-99e4c6b4dd68","resolution":{"observed_at":"2026-08-01T12:04:24.102817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-01T13:10:37.406289Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22716","last_updated":"2026-07-21T15:52:30Z","snapshot_observed_at":"2026-08-06T12:10:57.542923Z","submitted_at":"2026-07-21T15:52:30Z","title":"Visual Token Compression Enhances Robustness of MLLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T13:10:37.406289Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2607.22716"},"observation_digest":"sha256:c712eab39c0a1e4cff9d805b07d66b809c7bb19d33aded342cdb1fb2fcc7310a","observation_id":"37fc86a9-f4b7-4be8-b4bf-417450e1c4e0","resolution":{"observed_at":"2026-08-01T13:10:37.406289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2408.17003/citation-record","integrity":"/paper/2408.17003/integrity","json":"/paper/2408.17003/citation-record.json","paper":"/paper/2408.17003"},"outbound":[],"paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","latest_version":5,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 24 inbound Pith citation observations for arXiv:2408.17003."}