{"as_of":"2026-08-08T02:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9bd71dd6e1be3f16a0d4938a3a8fba53f58e30a75078a90593d2da692dc8bd38","coverage":[{"denominator":80,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":80,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T04:33:54.058475Z","state":"measured"},{"denominator":83,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":83,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T21:28:25.577739Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-06-30T21:35:04.761308Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"cited_work":{"arxiv_id":"2604.18519","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.18519","snapshot_observed_at":"2026-06-30T21:35:04.761308Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","venue":"cs.AI","work_id":"4ea0944a-8b8a-4722-a0f1-08c7293f8e4c","year":2026},"citing_paper":{"arxiv_id":"2605.06460","last_updated":"2026-05-07T15:51:06Z","snapshot_observed_at":"2026-07-06T23:18:55.724335Z","submitted_at":"2026-05-07T15:51:06Z","title":"MINER: Mining Multimodal Internal Representation for Efficient Retrieval","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-08T12:40:33.437364Z"},"links":{"cited_paper":"/paper/2604.18519","citing_paper":"/paper/2605.06460"},"observation_digest":"sha256:3552964322e10c6101209e473d8041a9f1a8289f61ec24c8b3281bb37343addd","observation_id":"85e0dbd5-f316-43eb-a156-6af8652f9f7f","resolution":{"observed_at":"2026-05-11T19:06:10.607800Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"cited_work":{"arxiv_id":"2604.18519","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.18519","snapshot_observed_at":"2026-06-30T21:35:04.761308Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","venue":"cs.AI","work_id":"4ea0944a-8b8a-4722-a0f1-08c7293f8e4c","year":2026},"citing_paper":{"arxiv_id":"2605.22786","last_updated":"2026-05-21T17:42:12Z","snapshot_observed_at":"2026-07-06T23:33:04.856017Z","submitted_at":"2026-05-21T17:42:12Z","title":"LCGuard: Latent Communication Guard for Safe KV Sharing in Multi-Agent Systems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T04:58:37.780195Z"},"links":{"cited_paper":"/paper/2604.18519","citing_paper":"/paper/2605.22786"},"observation_digest":"sha256:0e074b025e349f309fea12646cb2aa7f2e4b5b3282eafae5da54026a48180b78","observation_id":"89bd88d0-aa8e-4cca-80eb-23bae07202bb","resolution":{"observed_at":"2026-05-22T05:01:05.907482Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"cited_work":{"arxiv_id":"2604.18519","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.18519","snapshot_observed_at":"2026-06-30T21:35:04.761308Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","venue":"cs.AI","work_id":"4ea0944a-8b8a-4722-a0f1-08c7293f8e4c","year":2026},"citing_paper":{"arxiv_id":"2605.23974","last_updated":"2026-05-13T14:18:29Z","snapshot_observed_at":"2026-07-06T23:34:08.786038Z","submitted_at":"2026-05-13T14:18:29Z","title":"AERIC: Anticipatory Hidden-State Monitoring for Implicit Harmful Dialogue","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T21:28:25.577739Z"},"links":{"cited_paper":"/paper/2604.18519","citing_paper":"/paper/2605.23974"},"observation_digest":"sha256:f1c696874397d00960eebfdf3c7ccdbd309c331a76da3e9c1c16e2746e5d52bc","observation_id":"07051512-4efa-4e18-b1c7-73f64e217c15","resolution":{"observed_at":"2026-06-30T21:35:04.762580Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2604.18519/citation-record","integrity":"/paper/2604.18519/integrity","json":"/paper/2604.18519/citation-record.json","paper":"/paper/2604.18519"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:ca78d0ca39b397c7ec35b219093d48b754ab36b70f1a918bca0d081878a2bfd9","observation_id":"d5b2a56b-da7c-4d7e-99ac-83fb41c9db39","resolution":{"observed_at":"2026-05-10T12:20:23.136237Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T11:56:12.997668Z","title":"Proceedings of the AAAI conference on artificial intelligence , volume=","venue":null,"work_id":"ddd7b21f-c562-4dbb-a662-40884f12473c","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:a658f9ae0cccef3ce614f1507e80f1f5592c9592862fe6dd5f3497a1b623d48f","observation_id":"ca5b829a-b70c-420a-ab51-3c147be53ba6","resolution":{"observed_at":"2026-05-22T04:36:04.338928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05993","last_updated":"2024-09-11T14:42:29Z","snapshot_observed_at":"2026-08-06T16:17:24.580021Z","submitted_at":"2024-04-09T03:54:28Z","title":"AEGIS: Online Adaptive AI Content Safety Moderation with Ensemble of LLM Experts","version":2},"cited_work":{"arxiv_id":"2404.05993","doi":"10.48550/arxiv.2404.05993","metadata_source":"pith","pith_arxiv_id":"2404.05993","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AEGIS: Online Adaptive AI Content Safety Moderation with Ensemble of LLM Experts","venue":"cs.LG","work_id":"1ccd11c8-18f7-4764-989f-140f3f59c730","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2404.05993","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:da89a88567617bd0fcf6ae8963316780821ca5ddf7df8ad85c70c14445bbd12c","observation_id":"5d61a91f-2d50-47b3-a9f9-22c53dd92841","resolution":{"observed_at":"2026-05-10T12:20:23.127272Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09004","last_updated":"2025-01-15T18:37:08Z","snapshot_observed_at":"2026-08-08T01:23:43.364638Z","submitted_at":"2025-01-15T18:37:08Z","title":"Aegis2.0: A Diverse AI Safety Dataset and Risks Taxonomy for Alignment of LLM Guardrails","version":1},"cited_work":{"arxiv_id":"2501.09004","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.09004","snapshot_observed_at":"2026-07-03T10:17:58.098520Z","title":"AEGIS2.0: A diverse AI safety dataset and risks taxonomy for alignment of LLM guardrails","venue":null,"work_id":"d468418d-4f76-49bb-a0da-ebf21e257cbb","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2501.09004","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:d925c8548bc746cf0970f2dd0941c4db453edd00e6c23e05a6680245af3f6374","observation_id":"ef8cd7b4-e8c6-4fcc-a861-7dcac8db4b44","resolution":{"observed_at":"2026-05-10T12:20:23.119498Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.08370","last_updated":"2024-02-16T09:42:19Z","snapshot_observed_at":"2026-08-07T06:51:25.278756Z","submitted_at":"2023-11-14T18:33:43Z","title":"SimpleSafetyTests: a Test Suite for Identifying Critical Safety Risks in Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.08370","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.08370","snapshot_observed_at":"2026-07-04T03:59:32.652255Z","title":"arXiv preprint arXiv:2311.08370 , year=","venue":null,"work_id":"640cffcd-ef2c-4e9c-9585-872794809e9e","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2311.08370","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:2082612ad952339c3dc7bbf853413a4b60e197de414229e07f91b3d6578c31f0","observation_id":"3e4d273c-0bd7-403f-bafe-ec9e1246d42a","resolution":{"observed_at":"2026-05-10T12:20:23.111469Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04249","last_updated":"2024-02-27T04:43:08Z","snapshot_observed_at":"2026-07-06T17:26:23.067923Z","submitted_at":"2024-02-06T18:59:08Z","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","version":2},"cited_work":{"arxiv_id":"2402.04249","doi":"10.48550/arxiv.2402.04249","metadata_source":"pith","pith_arxiv_id":"2402.04249","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","venue":"cs.LG","work_id":"b0b0303f-2444-4789-a979-8153624312ff","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2402.04249","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:05075d728601638fb5b5f098892a914cd06943734af8f343cc1843811150d863","observation_id":"3f57f7dc-df1c-4f30-927d-2128b3986f66","resolution":{"observed_at":"2026-05-11T04:06:33.939395Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"7d497cfc-e2dd-45b9-b541-358cb97c4a09","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:075f50afb898efc47d18a2e5ae4ec210e4cdd805218664845762025536fe3497","observation_id":"d00c611a-d5db-4180-95cc-80bc559621db","resolution":{"observed_at":"2026-05-22T04:36:04.331178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-03T06:34:42.243765Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:f897a59671ce151ec5e61eca67d92846dc4f71441baeeafc1d903e8ba86d3841","observation_id":"1116b0a4-0e5d-4143-9f40-f09da8eb3e57","resolution":{"observed_at":"2026-05-10T12:20:23.122827Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T01:10:31.097671Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"35076cd8-2723-4c99-b604-65676abad5b8","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:24cbd895f4602160771b5bed892bff0752ae996c096b13e5ec45186f52694dc9","observation_id":"448b001b-4065-4a18-a656-900e9c4bfb9f","resolution":{"observed_at":"2026-05-22T04:36:04.335141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01263","last_updated":"2024-04-01T11:50:35Z","snapshot_observed_at":"2026-08-03T00:58:55.865010Z","submitted_at":"2023-08-02T16:30:40Z","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","version":3},"cited_work":{"arxiv_id":"2308.01263","doi":"10.48550/arxiv.2308.01263","metadata_source":"pith","pith_arxiv_id":"2308.01263","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"XSTest: A Test Suite for Identifying Exaggerated Safety Behaviours in Large Language Models","venue":"cs.CL","work_id":"bd953600-1547-4c1e-ade7-219a9f7cfe7a","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2308.01263","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:7ef607673c8171325f3720c158ceda342869cfca14910dd03174bcfa332b93db","observation_id":"6efb3c7b-b79d-4f91-a7e1-05a7c2bb9202","resolution":{"observed_at":"2026-05-15T06:51:50.894684Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.11878","last_updated":"2026-07-06T01:46:44Z","snapshot_observed_at":"2026-08-06T16:57:07.977935Z","submitted_at":"2025-07-16T03:48:03Z","title":"LLMs Encode Harmfulness and Refusal Separately","version":5},"cited_work":{"arxiv_id":"2507.11878","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.11878","snapshot_observed_at":"2026-07-07T03:18:08.851986Z","title":"Wenting Zhao, Xiang Ren, Jack Hessel, Claire Cardie, Yejin Choi, and Yuntian Deng","venue":null,"work_id":"cc7ef71b-dfa1-4d47-bb8e-6678c0827971","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2507.11878","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:b3a2056e3844752b871292a0085d8a9be75f5764d9a1712b01591226735e56da","observation_id":"e0afad29-862c-4db9-aa1b-a349345a1af3","resolution":{"observed_at":"2026-07-07T03:18:08.851986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"bee026ca-d43b-49ea-a45d-a04194b2206a","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:fb8e17c21d2a6a83650a0daf19a750aba7696db45f6631f2c5b4f505fa936fe2","observation_id":"ea42c941-7525-4501-9450-11bffdbb0f0b","resolution":{"observed_at":"2026-05-22T04:36:04.319477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.14276","last_updated":"2025-10-16T04:00:18Z","snapshot_observed_at":"2026-07-06T22:32:47.087967Z","submitted_at":"2025-10-16T04:00:18Z","title":"Qwen3Guard Technical Report","version":1},"cited_work":{"arxiv_id":"2510.14276","doi":"10.48550/arxiv.2510.14276","metadata_source":"pith","pith_arxiv_id":"2510.14276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3Guard Technical Report","venue":"cs.CL","work_id":"de1c5964-d29a-456c-97cc-30684b243a85","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2510.14276","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:5c3b9f44c3a2c6ae837cba7578e6ccc1863ac19d265af4d969ba6b07fb379bcd","observation_id":"7f1df501-f39a-4bbc-937e-38143a422378","resolution":{"observed_at":"2026-05-13T22:33:37.835217Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":"2407.04295","doi":"10.48550/arxiv.2407.04295","metadata_source":"pith","pith_arxiv_id":"2407.04295","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","venue":"cs.CR","work_id":"0ee7fc45-ae61-432b-83ac-f1d93ccd88fb","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2407.04295","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:67f67abf5c1c1a443c15c2dd4fb6af8230111cd21fd7703c3a0e7f298afecde8","observation_id":"84a353e3-dee7-49ab-9b6a-83160348045d","resolution":{"observed_at":"2026-05-15T02:20:44.922788Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-24T00:53:04.67981+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T00:53:04.67981+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18166","last_updated":"2024-06-14T07:27:26Z","snapshot_observed_at":"2026-07-06T18:21:18.150567Z","submitted_at":"2024-05-28T13:26:12Z","title":"Defending Large Language Models Against Jailbreak Attacks via Layer-specific Editing","version":2},"cited_work":{"arxiv_id":"2405.18166","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.18166","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Weiliang Zhao, Daniel Ben-Levi, Wei Hao, Junfeng Yang, and Chengzhi Mao","venue":null,"work_id":"8f894626-feb8-4c84-8100-eba286a00773","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2405.18166","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:f2b76288319f1231df84c27e86b4a255427f4b0079c3c35c8b375430bb049080","observation_id":"7645e6d1-bd18-4654-b0f6-c8e614faa764","resolution":{"observed_at":"2026-05-10T12:20:23.143551Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:51:39.975432Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"cc8c4cbe-6abe-4b13-96db-e4bad8c55202","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:a9008a3e069dc2e6aea004d9fd72bbb2b971db8b12c1ed82336df9b4f1904265","observation_id":"cc88ce6c-433a-4c59-9e53-59accdf3f005","resolution":{"observed_at":"2026-05-22T04:36:04.323469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T20:41:22.706830Z","title":"Proceedings of the 25th ACM SIGKDD international conference on knowledge discovery & data mining , pages=","venue":null,"work_id":"1aded522-8378-4d67-83a6-2988b961c09b","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:6acab7f3bf132b0950e8025f72fdc461e4011c4e8a093374b6b1783fec63e0b2","observation_id":"60ae356d-b6f3-4147-ae6d-213293efd60f","resolution":{"observed_at":"2026-05-22T04:36:04.327399Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Findings of the Association for Computational Linguistics: ACL 2024 , pages=","venue":null,"work_id":"e43154b5-e978-4307-8919-2e44d7662a7b","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:c27ff762bb9f1312074b2d423a772dc06652a8b78489703707687084a7d272c2","observation_id":"523891dd-59c8-4857-a9b1-cfa5e10863a8","resolution":{"observed_at":"2026-05-22T04:34:36.978394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The Eleventh International Conference on Learning Representations , year=","venue":null,"work_id":"fdfc8bdf-42af-4b1f-8897-369881081170","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:1080a6660f5bf8faf1c8e204003952e6bdc66f570e9d0fa84f7dc700890dfdb5","observation_id":"7ab56c69-546a-4139-a7be-3b6b496fbf07","resolution":{"observed_at":"2026-05-22T04:34:37.008478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T16:52:40.529973Z","title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":"be14d58e-a731-463c-9741-5c11489ead96","year":2021},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:cbe7d16d018699e78b52638dbcba4d51a7831dc8d1d951928c07e9379b99f489","observation_id":"1bd26b10-29c1-4ccf-a2be-4546be3cd448","resolution":{"observed_at":"2026-05-22T04:34:37.011054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.05950","last_updated":"2019-08-09T15:51:47Z","snapshot_observed_at":"2026-07-06T07:53:03.268985Z","submitted_at":"2019-05-15T05:47:23Z","title":"BERT Rediscovers the Classical NLP Pipeline","version":2},"cited_work":{"arxiv_id":"1905.05950","doi":"10.48550/arxiv.1905.05950","metadata_source":"arxiv_reference","pith_arxiv_id":"1905.05950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1905.05950 , year=","venue":"arXiv (Cornell University)","work_id":"47ec8f99-dad5-484e-bf9c-6fd2871df72b","year":2019},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/1905.05950","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:2d760febce0b7a3df980e39b41e635263fd5be2b574f77a906ed1b4959204bff","observation_id":"5a8ce3e2-c714-4813-9926-af1fd2a9d3b9","resolution":{"observed_at":"2026-05-10T12:20:23.051467Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":"eae7c9a6-9741-4149-bc1c-547f6241855f","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:88c4c81ff63fcf00fd2a166e4a90e65be1f6bf2e1b45cee94050c2b31e3a0695","observation_id":"b18c8fae-7c8a-41f2-8f99-48f1ad45b8c4","resolution":{"observed_at":"2026-05-22T04:34:36.932781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"physics/0004057","last_updated":"2000-04-24T15:22:30Z","snapshot_observed_at":"2026-08-07T10:46:08.274157Z","submitted_at":"2000-04-24T15:22:30Z","title":"The information bottleneck method","version":1},"cited_work":{"arxiv_id":"physics/0004057","doi":"10.48550/arxiv.physics/0004057","metadata_source":"pith","pith_arxiv_id":"physics/0004057","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The information bottleneck method","venue":"physics.data-an","work_id":"72655a80-0724-45ad-a330-1f4ed7aa613b","year":2000},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/physics/0004057","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:8722155d6e352dd0406091a1f1bb68e8c32a32295465dac5b7b64dc0967e7c23","observation_id":"43d98db9-d6d8-498c-bf60-daebb467c144","resolution":{"observed_at":"2026-05-10T12:20:23.068250Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-10T23:49:06.793856+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T23:49:06.793856+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T18:42:49.819176Z","title":"2022 , journal=","venue":null,"work_id":"81100a87-e8bd-4dc6-9857-2bc6713fbfa9","year":2022},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:dfba7ebaf6b53e0d5540dec3b402b65d5f5f94d0834b3c8452329a8b3520d17a","observation_id":"e34d1d47-aa1f-4d77-8861-8fe651e41ff1","resolution":{"observed_at":"2026-05-22T04:34:37.002119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transactions on Machine Learning Research , year=","venue":null,"work_id":"dfd6945c-adcc-459c-bd46-25cf435422c6","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:fc5377dee8ceb9ea1aa07559a4123223327333bbaca294fb1e20f9bfbd9d68a0","observation_id":"69f7e846-7b42-45c1-a132-6da1787dae30","resolution":{"observed_at":"2026-05-22T04:34:36.981356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2f7901a7-2bb2-4222-804a-a38a5ea1d854","year":2019},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:ba7f9578dcef7e83e0f281a94fd62f0312f11206ae0c4ec2c0d36b0703b996b4","observation_id":"69121aba-0375-44d3-ba70-9283af3100a9","resolution":{"observed_at":"2026-05-22T04:34:36.991231Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T08:06:58.158558Z","title":"Proceedings of the 2022 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":"d320b7c2-c576-4081-9326-6f4765a05468","year":2022},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:0894e0485f83691918444c334874dfb50dbd0dd5140666fc19cf88de82403b85","observation_id":"42bdf367-d710-42e8-8428-dcbef79212a7","resolution":{"observed_at":"2026-05-22T04:34:36.998239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"3d3a5284-5ff8-44d9-8cb2-15b849eb01e8","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:a9e58923dd90ba8a8eb4a638cf07f6001ae193413edc55a3494f09f6729ebbe1","observation_id":"5b69e8a9-a4ab-4deb-abb0-dd3659881546","resolution":{"observed_at":"2026-05-22T04:34:36.987886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02013","last_updated":"2025-06-15T18:27:17Z","snapshot_observed_at":"2026-08-01T16:50:50.645857Z","submitted_at":"2025-02-04T05:03:42Z","title":"Layer by Layer: Uncovering Hidden Representations in Language Models","version":2},"cited_work":{"arxiv_id":"2502.02013","doi":"10.48550/arxiv.2502.02013","metadata_source":"pith","pith_arxiv_id":"2502.02013","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Layer by Layer: Uncovering Hidden Representations in Language Models","venue":"cs.LG","work_id":"7b4ac06a-e804-4f0a-8305-c45f2735afb5","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2502.02013","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:91f17b14699c511b6a7805a970d2cb7c78f79b132048e96a958cd5c5e75c6851","observation_id":"a4b800ca-a3da-43b0-a7b1-a1fda654f7fb","resolution":{"observed_at":"2026-05-15T16:30:37.615120Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06674","last_updated":"2023-12-07T19:40:50Z","snapshot_observed_at":"2026-07-06T17:00:00.321552Z","submitted_at":"2023-12-07T19:40:50Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","version":1},"cited_work":{"arxiv_id":"2312.06674","doi":"10.3390/info16050365","metadata_source":"pith","pith_arxiv_id":"2312.06674","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","venue":"cs.CL","work_id":"93844332-869b-448c-a1be-35466150b1b2","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2312.06674","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:f4955b2f666092c82217f11eee52d312d8b06fb3aeca02a95c3b3f89a9a19ad1","observation_id":"0184305d-8bb6-4eff-83ea-c80f32d7ecf2","resolution":{"observed_at":"2026-05-10T18:59:26.864775Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Organization of Knowledge and Advanced Technologies","venue":null,"work_id":"b729d205-8554-4c88-81c0-36636e80a272","year":2020},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:cf18a9529ac66597a2720dfe14fc097f1d550ac3c64320bde8912789795eac1e","observation_id":"f26b44fe-ecee-4a67-9ae5-c4390c5e1a6a","resolution":{"observed_at":"2026-05-22T04:34:36.994809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16174","last_updated":"2026-06-01T15:28:47Z","snapshot_observed_at":"2026-08-07T17:56:39.289394Z","submitted_at":"2025-02-22T10:31:50Z","title":"Efficient LLM Moderation with Multi-Layer Latent Prototypes","version":4},"cited_work":{"arxiv_id":"2502.16174","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.16174","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient llm moderation with multi-layer latent prototypes","venue":null,"work_id":"ee9aa5d6-1ec4-4e9a-b723-5a7a6c7a7789","year":2026},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2502.16174","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:0095e0104c0b2461da9d866e682f91820f14c8720636404c39c2a71a61705c9f","observation_id":"8fec285d-6f79-4c4b-af1c-b823e1b66dcf","resolution":{"observed_at":"2026-06-02T04:04:22.908302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":"2408.17003","doi":"10.48550/arxiv.2408.17003","metadata_source":"arxiv_reference","pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety layers of aligned large language models: The key to llm security","venue":"arXiv (Cornell University)","work_id":"4807e5d2-329e-4c63-bbe1-67f643b30f78","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:647256b7564c18d31e0d23e779ddc6261ca7f7aac6880ab8ed10725539c78a4d","observation_id":"d09d643a-6e56-4864-a8ff-6dafdace5ff8","resolution":{"observed_at":"2026-05-10T12:20:23.070753Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":"b64fb00e-4d22-4c08-961b-69d1a2339181","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:d40f2350a0dba306fb7f8be0f12865e06314c86e92d5fac09b94123f42e48f9a","observation_id":"c21951ba-222c-4ee3-8d29-9d89bd632016","resolution":{"observed_at":"2026-05-22T04:34:36.984404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.06594","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:40:01.019567Z","title":"arXiv preprint arXiv:2510.06594 , year=","venue":null,"work_id":"c94b177c-9e46-4186-b64b-83f11e965c8e","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:6595587d90544f62ae94308d0c97a8a2444c68435f345981ca2d61193357d0d9","observation_id":"288a5fb8-4d8c-41d7-add8-f2f1fab2587d","resolution":{"observed_at":"2026-05-10T12:20:23.079781Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09124","last_updated":"2024-02-15T19:12:10Z","snapshot_observed_at":"2026-08-07T19:01:56.708259Z","submitted_at":"2023-08-17T17:59:19Z","title":"Linearity of Relation Decoding in Transformer Language Models","version":2},"cited_work":{"arxiv_id":"2308.09124","doi":"10.48550/arxiv.2308.09124","metadata_source":"pith","pith_arxiv_id":"2308.09124","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Linearity of relation decoding in transformer language models","venue":"cs.CL","work_id":"25f5f724-b6d7-427f-a2f3-e8b72fd3b5e2","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2308.09124","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:24d3bb7286e78cf482ed55b475b961c44abb044ac7151697bd325c516cb3dd26","observation_id":"1935def0-d11b-4f5c-8a33-228208403b18","resolution":{"observed_at":"2026-05-10T12:20:23.037529Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08112","last_updated":"2025-11-11T01:13:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-14T17:47:09Z","title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","version":6},"cited_work":{"arxiv_id":"2303.08112","doi":"10.48550/arxiv.2303.08112","metadata_source":"pith","pith_arxiv_id":"2303.08112","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","venue":"cs.LG","work_id":"a127314f-7424-488f-b6d7-8214650c420f","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2303.08112","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:b9295060b283957204409b96c4de010f455c6f918e7c5a25474ec484d37f8b1c","observation_id":"da6ffd10-3a93-4014-835a-f3291df9f26e","resolution":{"observed_at":"2026-05-12T16:54:37.831311Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-06-01T22:57:41.570848+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T22:57:41.570848+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Companion Proceedings of the ACM on Web Conference 2025 , pages=","venue":null,"work_id":"aaf9bf1e-c904-40c6-9fdf-458c5fde179b","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:6fe4730e6c4be4604c0a1cb6b8fe9e22dbee41c95036a7e205beeed5dbda06ee","observation_id":"1a266207-ff1e-4b3c-a046-ba8b81cf6fb0","resolution":{"observed_at":"2026-05-22T04:34:37.005418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Findings of the Association for Computational Linguistics: ACL 2025 , pages=","venue":null,"work_id":"c623ad96-b9c9-496c-a2b2-2f1ae578fe13","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:c6593dbfa4efc27655fac78453a6652941fde5220a3a79fcc61ae596070aebd2","observation_id":"9a5eb6c6-89f2-4c7a-ae93-07f870c29af0","resolution":{"observed_at":"2026-05-22T04:34:36.941560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03658","last_updated":"2024-07-17T22:24:27Z","snapshot_observed_at":"2026-07-06T16:43:58.947915Z","submitted_at":"2023-11-07T01:59:11Z","title":"The Linear Representation Hypothesis and the Geometry of Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.03658","doi":"10.48550/arxiv.2311.03658","metadata_source":"pith","pith_arxiv_id":"2311.03658","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Linear Representation Hypothesis and the Geometry of Large Language Models","venue":"cs.CL","work_id":"a7b44adc-f2c2-4420-a27d-8ade97dd3b75","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2311.03658","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:95fcd27f4a4621bd2f4a3c17ad305dfdebfc023f33b683758ebd11897780b47c","observation_id":"95ba164b-e9b8-4273-83dc-726ed179b39c","resolution":{"observed_at":"2026-05-11T21:43:32.339354Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-18T08:21:05.185759+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T08:21:05.185759+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Understanding intermediate layers using linear classifier probes , url =","venue":null,"work_id":"343d1f02-0415-44dd-8995-d78e6518bf55","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:c1945dd009cb40bc9c5d3fbc309382fcacba22c977f8f3329e4c230ad4d914c7","observation_id":"fa7e2d14-1c43-4d50-8740-2df308cfcdac","resolution":{"observed_at":"2026-05-22T04:34:36.958245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An introduction to variable and feature selection , url =","venue":null,"work_id":"a8a2b7d9-7cd9-4bc0-a6d8-82b647612060","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:6c180920226c34517e4031cfd7d15063831995551fcf39fef4fea7755c29b22e","observation_id":"af32c80a-6a32-45b6-a975-6977ccb54369","resolution":{"observed_at":"2026-05-22T04:34:36.938925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:2b6f3c608cb5ec404b63601716cab61194ccbbd189075d5435c75fef88d68916","observation_id":"ec5f792e-2537-4fb4-a19e-bd90e7521df7","resolution":{"observed_at":"2026-05-10T12:20:23.019486Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T22:54:10.690500Z","title":"arXiv e-prints , pages=","venue":null,"work_id":"e3d36c3d-7b21-4239-8c39-24564d8e9696","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:85b903a0d9bc68287f062d89bbf5f46a739a2d8bd514bc7e02068ccd77acfb4c","observation_id":"aa71f58b-386e-47f6-9ee3-b2b19877f419","resolution":{"observed_at":"2026-05-22T04:34:37.013438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13435","last_updated":"2024-12-18T02:13:13Z","snapshot_observed_at":"2026-07-06T20:08:56.278699Z","submitted_at":"2024-12-18T02:13:13Z","title":"Lightweight Safety Classification Using Pruned Language Models","version":1},"cited_work":{"arxiv_id":"2412.13435","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13435","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sawtell, T","venue":null,"work_id":"e0768f99-e436-4698-8849-7219d400a149","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2412.13435","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:f3306df553b27cb668f29fa63711840ebc14a45475b2da3d2340dcff6cdce401","observation_id":"9b434071-60ab-45dd-94a1-f0891d6c830a","resolution":{"observed_at":"2026-05-10T12:20:23.025851Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"cited_work":{"arxiv_id":"2407.21772","doi":"10.48550/arxiv.2407.21772","metadata_source":"pith","pith_arxiv_id":"2407.21772","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","venue":"cs.CL","work_id":"6d0d9d39-490d-48d5-b346-6bfe40b8b8fb","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2407.21772","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:0722bc618b7c9eb5dd7526c9111987b72d0c108e55c9620e3e988db8302a6ab0","observation_id":"37f1d075-d1a1-44e1-bf1c-d313c220631a","resolution":{"observed_at":"2026-05-20T13:17:39.551920Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.emnlp-main.486","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Generative or Discriminative? Revisiting Text Classification in the Era of Transformers","venue":null,"work_id":"9c993cf3-4845-47b4-8420-1dfd1b417445","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:276e3810ac31af4ccc8c2b9c31f913b9daadc9d4fe513aa0056752178239c083","observation_id":"ed317d6a-02e1-4c85-9ecf-89b825591580","resolution":{"observed_at":"2026-05-10T04:35:16.476296Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The Thirteenth International Conference on Learning Representations , year=","venue":null,"work_id":"a941a4e3-1686-43c1-ad52-d2de098240dc","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:0b44bd0bf97287b2d3150541114cd078824cf4d52620b67864f862c90f067aff","observation_id":"4dd6b657-03f0-4045-9038-4df151e4383b","resolution":{"observed_at":"2026-05-22T04:34:37.023071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T16:42:40.001878Z","title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":"5a7e6354-0582-413c-a0b6-ed5573b425de","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:ec9bc898e3f921a7ef8a60709de71c9520d444e2e66a4d945f1ff81149a4a9f8","observation_id":"9e0f907f-bdd2-42f2-8de5-68a3b7994bf9","resolution":{"observed_at":"2026-05-22T04:34:36.948506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE Robotics and Automation Letters , volume=","venue":null,"work_id":"2b5566de-5eb1-4dcc-9642-2abb0cab9de2","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:f9b70d8f1864b2ecc896399a23ba55430b70da28941f6276bd816dffd6c77b4b","observation_id":"134b24df-8dfe-496c-a19f-abc97a7ff2f8","resolution":{"observed_at":"2026-05-22T04:34:36.978073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18556","last_updated":"2025-05-26T03:00:39Z","snapshot_observed_at":"2026-08-07T16:01:52.313351Z","submitted_at":"2025-04-16T10:05:37Z","title":"RDI: An adversarial robustness evaluation metric for deep neural networks based on model statistical features","version":2},"cited_work":{"arxiv_id":"2504.18556","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18556","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2504.18556 , year=","venue":null,"work_id":"afed160d-49b3-4684-97e8-a8d9c2972a97","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2504.18556","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:0c08e64d6beab53cef52b87c01e065bc7491ba3a266e97d5c135bb9495d7e223","observation_id":"dff3646d-ff43-490f-af8f-75e286bb77b7","resolution":{"observed_at":"2026-05-10T12:20:23.060041Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T06:50:48.645747Z","title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) , pages=","venue":null,"work_id":"3c05557c-2471-4d17-b026-259d07bff219","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:4724e7b29a1dcb59cd4f90cca46d2b27728f8af1cf7b74e981b1c543cf531399","observation_id":"0352f629-72d9-4450-8d61-fe393b41a693","resolution":{"observed_at":"2026-05-22T04:34:36.944899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling laws for neural language models , url =","venue":null,"work_id":"5fd4078c-4c55-4448-8124-fb5a761459a5","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:d22f3fcfa66173d3fdc607be800e79b4af8706434b73143f17cf78d5edf4c7ea","observation_id":"25396c11-0073-4387-9657-1dcad9e06309","resolution":{"observed_at":"2026-05-22T04:34:37.031134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:996605ed69ead4eb2d7cc574b3ea39fa15b2abfd8c68090d94b1e690e4cae7ad","observation_id":"ecedd5a7-16f4-4a84-b195-84538c5d991d","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T15:16:18.791332Z","title":"author=","venue":null,"work_id":"78d5b8e1-1042-4222-9246-42cfa74bf034","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:777f2ddd84925fa67f9c1996eb8c201b1cbf0655d5c4f188325b5ba28bd3ea49","observation_id":"4b56e116-1bfb-49f2-85bf-d149403d1cdf","resolution":{"observed_at":"2026-05-22T04:34:37.027719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":"2307.15043","doi":"10.48550/arxiv.2307.15043","metadata_source":"pith","pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","venue":"cs.CL","work_id":"3322fa86-1768-4677-8425-dd326b45e078","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:84917fec681cc6921076c4d3a47059c174a68884de39f5ec6ef99394630ae71e","observation_id":"f437287a-2719-4f3a-9179-e74fe41ecda4","resolution":{"observed_at":"2026-05-10T12:20:23.010454Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:40.271168+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:40.271168+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T17:31:22.428029Z","title":"2025 , howpublished =","venue":null,"work_id":"e15f2e3c-ed7d-4865-9622-6db1c35b3787","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:081f46b1a97169b1319a1d33cb75da1f57fe84971d2354d16cdb400282e526ba","observation_id":"0c8b0a62-c9a2-44f9-b6b3-d4a9fbeb5989","resolution":{"observed_at":"2026-05-22T04:36:04.302967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T06:42:00.895480Z","title":"2025 , howpublished =","venue":null,"work_id":"cf449263-d6d6-4ea4-bc59-d85c8e3b940e","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:0325d2abc9a274c6c483a4cf53c9c20c8a405d93ce10dc7293293b4199ab2d2f","observation_id":"9e552a06-6d22-4761-a2d0-2b53260aa24c","resolution":{"observed_at":"2026-05-22T04:34:36.951802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T19:27:32.561811Z","title":null,"venue":null,"work_id":"48a06fba-cfc0-4db3-9a8d-e2a4d8c69eb9","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:f2a4f32ba67dac60025b3404983f745dbbbe4b8984295158d160a7278ae8aee9","observation_id":"7b9dbb1e-d773-4a52-ac22-4b31eac9f159","resolution":{"observed_at":"2026-05-22T04:34:36.955330Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02727","last_updated":"2024-09-05T07:17:59Z","snapshot_observed_at":"2026-07-06T19:10:25.534956Z","submitted_at":"2024-09-04T14:01:48Z","title":"Pooling And Attention: What Are Effective Designs For LLM-Based Embedding Models?","version":2},"cited_work":{"arxiv_id":"2409.02727","doi":"10.48550/arxiv.2409.02727","metadata_source":"arxiv_reference","pith_arxiv_id":"2409.02727","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Ł ukasz Kaiser, and Illia Polosukhin","venue":"arXiv (Cornell University)","work_id":"d72d008e-2950-473b-b50a-967836bff69a","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2409.02727","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:1319f1eeda0a1cd53645c273923fd4dff70facc8d57e24e1beec0c4648d52e8f","observation_id":"4ffef57a-39c4-491a-8344-0b79e4db3752","resolution":{"observed_at":"2026-05-10T12:20:23.048764Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:8e7812e89ea244dd546e9ab0a6955b5e8253658aad9f4b3f7364b5196e624a63","observation_id":"7426ea55-99ba-44c8-a9e5-397ca6e2f056","resolution":{"observed_at":"2026-05-10T12:20:23.040333Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":"2406.12793","doi":"10.48550/arxiv.2406.12793","metadata_source":"pith","pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","venue":"cs.CL","work_id":"de9ce5af-0d8d-4b94-9793-64968d9bc06d","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:d383c66df2a2e1ec49b9a35dc2a980764cc44852f78b1e006e1ca408b0cc44d2","observation_id":"96928c83-816f-4f73-adf8-bb0302d28415","resolution":{"observed_at":"2026-05-11T08:08:10.048088Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.03550","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2508.03550 , year=","venue":null,"work_id":"212eb3dc-26a3-42e4-adad-95df6b27bfe6","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:776bd8547f304f56512cb5fbd19837610dc5c1dcea2077b4955a4a35b16d0dec","observation_id":"190f9fcb-274b-4fad-91b2-00244340b913","resolution":{"observed_at":"2026-05-10T12:20:23.045949Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"d3f9a893-7df5-46dc-88ba-a81ecf3a309d","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:bbeb8f40f4674fc868ac7aaf40b6f83d9e6009d83524dc0cd7a171e05070449d","observation_id":"c401dee2-40a4-4c16-a905-26cd98ef813d","resolution":{"observed_at":"2026-05-22T04:36:04.294906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n19-1423","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BERT : Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"Proceedings of the 2019 Conference of the North","work_id":"3e3c8ac8-b858-4b22-af32-393d98c883e0","year":2019},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:a7c8bebb67c7e2e7a5f64c9e1c736325f02659765836a04955fab999b9e43ad6","observation_id":"fd9f2d22-03f3-4eac-a64c-2d087444a635","resolution":{"observed_at":"2026-05-10T04:35:16.486832Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-01T13:38:13.85894+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T13:38:13.85894+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T17:16:24.057499Z","title":"2019 , eprint=","venue":null,"work_id":"7a564fc3-fda4-49f2-9c6b-10c653f76e47","year":2019},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:fca22dda9ec4b1182bc0828979931a84dc0913dbdda25d5af4a3b19a0d906fd9","observation_id":"5b760b1f-10fb-4087-a3cf-f513f2d6604f","resolution":{"observed_at":"2026-05-22T04:36:04.298967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1371/journal.pone.0237861","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Hate speech detection and racial bias mitigation in social media based on bert model.PLOS ONE, 15(8):1–26, 08 2020","venue":"PLoS ONE","work_id":"aeee4220-f8e7-4719-b198-a2e2f1dbc3b5","year":2020},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:1d70dfadbef6a1a032d1b64e067e3a93046b89110a441aba49b842d34a177d23","observation_id":"d0c1504c-2ded-4914-9a9f-f6a8b85423c4","resolution":{"observed_at":"2026-05-10T04:35:16.482488Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2442.345231","doi":"10.1145/3442442.3452313","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2021 , isbn =","venue":"Companion Proceedings of the Web Conference 2021","work_id":"4d15dc54-5257-4da4-9b44-38a7b5fa274b","year":2021},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:87453c4cca8fad8e709374b7cb9ef7f0cefa7d3185213313dcec9daa57f4a05a","observation_id":"cc4b602a-2457-4275-b6ea-74d178d22fdf","resolution":{"observed_at":"2026-05-10T04:35:16.480828Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.woah-1.3","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"H ate BERT : Retraining BERT for Abusive Language Detection in E nglish","venue":null,"work_id":"2bcfea56-618e-4a3e-b76f-d2eb12b37f92","year":2021},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:a751fd7f7a60c2dbc1fbc2b211361ab65c90c847f469f2779808d92b73074c97","observation_id":"10d91182-2d94-4175-8282-2944ecafb995","resolution":{"observed_at":"2026-05-10T04:35:16.478300Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T22:57:43.199137Z","title":"2024 , eprint=","venue":null,"work_id":"56de79bb-a384-4567-8274-19e8bfa69568","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:1f34e8d97aff7be0ee69c6e6833864bf1412a29968443d887b96d9a6a9531e91","observation_id":"02b07d0d-af88-48fd-8079-717a37ffa34a","resolution":{"observed_at":"2026-05-22T04:36:04.306299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4678.353914","doi":"10.1145/3534678.3539140","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Tay, Yi and Sorensen, Jeffrey and Gupta, Jai and Metzler, Donald and Vasserman, Lucy , title =","venue":"Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","work_id":"c02a9b27-3564-4357-915c-e7d44f32a29c","year":2022},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:81331c7009f8e2849fa1df35d4b0a4d0c9f1de7213c6d697e99f4a98a44572e4","observation_id":"06a6ec07-824a-4b46-8aa8-5c30d572501c","resolution":{"observed_at":"2026-05-10T04:35:16.485012Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T20:25:37.838480Z","title":"2025 , eprint=","venue":null,"work_id":"33e0cb0b-704f-4a42-a021-a81471af4087","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:b6c8eba2a1bc3e7ad9f3ae5cb9cec8857bd6fe3075375522a02064293168b1b5","observation_id":"5abbd1ea-32e1-4861-b346-9b5e7c58d9c5","resolution":{"observed_at":"2026-05-22T04:34:37.034244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems , year=","venue":null,"work_id":"cd271955-39e9-462d-ba10-ea10e08b7d30","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:18639553879646846c60ec8b7b125129b89f693794bb04f1805ea78e8cb55235","observation_id":"0421a101-051c-442a-a6b4-4b3a122c85d1","resolution":{"observed_at":"2026-05-22T04:36:04.309711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04377","last_updated":"2025-08-07T14:05:19Z","snapshot_observed_at":"2026-08-07T16:08:16.694760Z","submitted_at":"2025-04-06T06:09:21Z","title":"PolyGuard: A Multilingual Safety Moderation Tool for 17 Languages","version":2},"cited_work":{"arxiv_id":"2504.04377","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.04377","snapshot_observed_at":"2026-07-08T10:04:51.407053Z","title":"Polyguard: A multilingual safety moderation tool for 17 lan- guages","venue":"cs.CL","work_id":"463b6c7a-05d7-46af-8cec-d314068a7e1b","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2504.04377","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:d8bb4537a58b137bb38cd03f2ff8c58e9bb51f4fa983f083746db073309c08bb","observation_id":"1e9acd6d-91b0-4870-a830-94de92d2b6cd","resolution":{"observed_at":"2026-05-10T12:20:23.043286Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15154","last_updated":"2023-10-23T17:55:31Z","snapshot_observed_at":"2026-08-08T01:44:57.952179Z","submitted_at":"2023-10-23T17:55:31Z","title":"Linear Representations of Sentiment in Large Language Models","version":1},"cited_work":{"arxiv_id":"2310.15154","doi":"10.48550/arxiv.2310.15154","metadata_source":"pith","pith_arxiv_id":"2310.15154","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Linear Representations of Sentiment in Large Language Models","venue":"cs.LG","work_id":"6cb3c7a7-3301-449f-97b9-7e047edafdf9","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.15154","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:cc4c1edd2a80143c42b5a31c44b457b81a50d4d31d1190842768b9945c062976","observation_id":"1a853100-0621-4b91-8431-916beeb54807","resolution":{"observed_at":"2026-05-15T12:43:06.705954Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06824","last_updated":"2024-08-19T01:18:41Z","snapshot_observed_at":"2026-07-06T16:30:37.867641Z","submitted_at":"2023-10-10T17:54:39Z","title":"The Geometry of Truth: Emergent Linear Structure in Large Language Model Representations of True/False Datasets","version":3},"cited_work":{"arxiv_id":"2310.06824","doi":"10.18653/v1/2025.findings-acl.38","metadata_source":"pith","pith_arxiv_id":"2310.06824","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Geometry of Truth: Emergent Linear Structure in Large Language Model Representations of True/False Datasets","venue":"cs.AI","work_id":"400e017f-8643-4166-b6da-a75d4446da80","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.06824","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:48948d591b93797ad99e08171b39f1e24bd481f89d72f25eca5812b145f0de3a","observation_id":"36b3f119-44c3-4de0-8410-ade08be64a1a","resolution":{"observed_at":"2026-05-12T19:35:39.400821Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the 28th ACM international conference on information and knowledge management , pages=","venue":null,"work_id":"6cebc7d8-ad35-4ec4-b3c9-e621426622d9","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:c5e3caffd6c5cb4e41b637f18d80e5a78dc04a3ae5e29d5e174dfca8f468b550","observation_id":"279548d1-4dae-4d91-9093-32759b537cbf","resolution":{"observed_at":"2026-05-22T04:36:04.315979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.18081","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2510.18081 , year=","venue":null,"work_id":"dfd3a9df-2167-45e2-aa78-c6cff685ad2e","year":null},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:1f9eb4474aca47eda3da20370118091044ff8441a6102f6596ebbbeaf9982859","observation_id":"f2aa0f24-f7bf-4c7f-9eb9-2c042ce9d75b","resolution":{"observed_at":"2026-05-10T12:20:23.065573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.03502","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Curvalid: Geometrically-guided adversarial prompt detection","venue":null,"work_id":"decd5ea9-0cf9-4b81-8a0b-1e60b0fc7d3e","year":2025},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:c804b1d612585250cd32338f01b641e014a8c55dfe886af78aa612a8215b0adc","observation_id":"9b0a7544-a00d-4285-9474-24c935d534b5","resolution":{"observed_at":"2026-05-10T12:20:23.073944Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T05:06:48.347155Z","title":"Neural networks , volume=","venue":null,"work_id":"c09ae318-7fb2-44c0-97cc-886c5f62e65b","year":1992},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:4065ed0888588d271247d5d7ef9b1bbce7e603942d57855747dfdbcab6597dc7","observation_id":"79fa70d3-2184-45cb-b67e-8ed50c7fd44f","resolution":{"observed_at":"2026-05-22T04:36:04.312630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations"},"reference_resolution":{"displayed":80,"state_counts":{"malformed_identifier":0,"metadata_mismatch":30,"parse_uncertain":1,"unresolved":1,"verified_exact":12,"verified_fuzzy":36},"total_outbound_references":80},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 80 of 80 outbound references and 3 inbound Pith citation observations for arXiv:2604.18519."}