{"as_of":"2026-08-08T04:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f2502828110967c5d59495716b7c47fc13a29d6e8e29f689e20f261615b200db","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:52:28.909375Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T14:48:32.654187Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2406.18495","last_updated":"2024-12-09T20:21:56Z","snapshot_observed_at":"2026-08-06T04:32:59.039501Z","submitted_at":"2024-06-26T16:58:20Z","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T16:25:14.744887Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2406.18495"},"observation_digest":"sha256:ba0b831d384c479a8df338336b3822f19af3a12c923f745b515f674bceceafd8","observation_id":"4ae480b6-9c61-470c-a90d-6399246e790d","resolution":{"observed_at":"2026-05-17T16:25:14.866587Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-20T13:17:39.444002Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2407.21772"},"observation_digest":"sha256:1deb7d255d218ee337e8b6ee1f81f975efc2b97abc00ffdf0ebf52abeae853ab","observation_id":"a6049c2b-aeaa-44a8-8685-d434ac56dbb4","resolution":{"observed_at":"2026-05-20T13:17:39.523130Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T14:52:28.909375Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-07T23:00:04.573369Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.909375Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:82036c0af1fa3c865373cd86edc577e533465fdaa308c3a67c82014d9fdf63ab","observation_id":"e773cb32-b65a-4db0-8695-1cd4132b6e08","resolution":{"observed_at":"2026-08-07T14:52:28.909375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T13:16:15.795060Z","title":"Thomas Hartvigsen, Saadia Gabriel, Hamid Palangi, Maarten Sap, Dipankar Ray, and Ece Kamar","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23836","last_updated":"2025-07-16T11:25:40Z","snapshot_observed_at":"2026-08-07T13:08:50.821464Z","submitted_at":"2025-05-28T12:03:09Z","title":"Large Language Models Often Know When They Are Being Evaluated","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T13:16:15.795060Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2505.23836"},"observation_digest":"sha256:4bdebc64aaf034cc6774ae5fbb7946573b84e6bfbd1d898cb773f17c16f5f2d5","observation_id":"96d8ff20-356b-4be2-9aba-bb9b38955dea","resolution":{"observed_at":"2026-08-07T13:16:15.795060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T10:28:54.802416Z","title":"Lin et al., ‘ToxicChat: Unveiling Hid- den Challenges of Toxicity Detection in Real-World User-AI Conversation’, Oct","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06391","last_updated":"2025-06-05T16:53:29Z","snapshot_observed_at":"2026-08-08T01:20:47.850082Z","submitted_at":"2025-06-05T16:53:29Z","title":"From Rogue to Safe AI: The Role of Explicit Refusals in Aligning LLMs with International Humanitarian Law","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:54.802416Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2506.06391"},"observation_digest":"sha256:48e3342fbb74225da5d9b68a51723713e66a9e7bdfaebdda17d90d3b6528ea7c","observation_id":"c0f879a3-92f5-496f-abec-9b8f05638b4d","resolution":{"observed_at":"2026-08-07T10:28:54.802416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-07T10:17:26.647489Z","title":"Toxicchat: Unveiling hidden challenges of toxicity detection in real- world user-ai conversation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.647489Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:3b00593451c8669d2cbe2413b31f96c2f8515d2b9813791e43795e6cd3897c8f","observation_id":"899debdd-8dbc-4652-b50f-8fe1d3245bd9","resolution":{"observed_at":"2026-08-07T10:17:26.647489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-06T21:45:02.699513Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation , 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23706","last_updated":"2025-06-30T10:29:42Z","snapshot_observed_at":"2026-08-06T21:31:17.495341Z","submitted_at":"2025-06-30T10:29:42Z","title":"Attestable Audits: Verifiable AI Safety Benchmarks Using Trusted Execution Environments","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T21:45:02.699513Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2506.23706"},"observation_digest":"sha256:dc7a802f50d842cd34b66fcb65946ef374895edfe227b625c1aea7f8e7514dd6","observation_id":"a59b22d2-ab12-433d-8fca-fcf93f1968f3","resolution":{"observed_at":"2026-08-06T21:45:02.699513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-06T17:03:15.999906Z","title":"Toxicchat: Unveiling hidden challenges of toxicity detection in real-world user-ai conversation","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11878","last_updated":"2026-07-06T01:46:44Z","snapshot_observed_at":"2026-08-06T16:57:07.977935Z","submitted_at":"2025-07-16T03:48:03Z","title":"LLMs Encode Harmfulness and Refusal Separately","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:03:15.999906Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2507.11878"},"observation_digest":"sha256:247796c6e56ab771d68acb786e5cf86718a523b3eeb57276cb607f3fcc5ba81d","observation_id":"8945f4db-2a41-4c61-9a11-ef0ac1d66d3e","resolution":{"observed_at":"2026-08-06T17:03:15.999906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2510.13727","last_updated":"2026-05-19T04:39:26Z","snapshot_observed_at":"2026-08-02T19:37:00.796741Z","submitted_at":"2025-10-15T16:30:57Z","title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T20:42:40.823721Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2510.13727"},"observation_digest":"sha256:914608930615cd232537949239de4af90ebb686112843336a89bdb826daa7156","observation_id":"7d08457f-5a13-4da3-af0c-6a428b8406fa","resolution":{"observed_at":"2026-05-21T20:44:22.044971Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-03T05:07:09.235559Z","title":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.03328","last_updated":"2026-05-27T13:01:56Z","snapshot_observed_at":"2026-08-04T00:07:19.801811Z","submitted_at":"2026-02-03T09:56:20Z","title":"GuardReasoner-Omni: A Reasoning-based Multi-modal Guardrail for Text, Image, Video, and Audio","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T05:07:09.235559Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2602.03328"},"observation_digest":"sha256:ac90c3d8529d83de6d4b2391fc5c8dea4fccedd85acd4525aa0ddc2925befde2","observation_id":"fe7c1cb3-98e9-4202-af1b-118f250474bf","resolution":{"observed_at":"2026-08-03T05:07:09.235559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-03T01:17:12.009201Z","title":"Toxicchat: Unveiling hidden challenges of toxicity detection in real-world user-ai conver- sation.arXiv preprint arXiv:2310.17389, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.10388","last_updated":"2026-05-29T05:05:29Z","snapshot_observed_at":"2026-08-07T17:40:15.715656Z","submitted_at":"2026-02-11T00:23:13Z","title":"Less is Enough: Synthesizing Diverse Data in LLM Feature Space with Sparse Autoencoders","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-03T01:17:12.009201Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2602.10388"},"observation_digest":"sha256:e191c895c155f6d21af736182044b30f7d4a2b037dcb10ee197f1b5bcdf9678d","observation_id":"86848d71-d41c-4229-aecd-7ea9e6c58cbc","resolution":{"observed_at":"2026-08-03T01:17:12.009201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.06833","last_updated":"2026-04-08T08:51:46Z","snapshot_observed_at":"2026-07-06T22:55:15.791334Z","submitted_at":"2026-04-08T08:51:46Z","title":"FedDetox: Robust Federated SLM Alignment via On-Device Data Sanitization","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T17:22:40.613937Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.06833"},"observation_digest":"sha256:d6c81a2e5c2e03660839403c6dd069073668efbb8a5ed78ab52738ed75a8631a","observation_id":"f7e3cf6a-63e5-4fe4-b5bc-3396ff7d194c","resolution":{"observed_at":"2026-05-11T06:56:01.782343Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.07655","last_updated":"2026-04-08T23:47:29Z","snapshot_observed_at":"2026-08-02T09:41:13.361711Z","submitted_at":"2026-04-08T23:47:29Z","title":"Guardian-as-an-Advisor: Advancing Next-Generation Guardian Models for Trustworthy LLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T17:27:13.339411Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.07655"},"observation_digest":"sha256:225bb77b26fcd2f5735dce3dd4bfe92fcee726c63d367e450ee465561fc364fe","observation_id":"14c65156-fe1a-4726-8f6c-d6fff637a2ff","resolution":{"observed_at":"2026-05-11T06:46:37.658425Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.16542","last_updated":"2026-04-17T01:55:37Z","snapshot_observed_at":"2026-08-01T22:59:54.227625Z","submitted_at":"2026-04-17T01:55:37Z","title":"TWGuard: A Case Study of LLM Safety Guardrails for Localized Linguistic Contexts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T09:07:57.713675Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.16542"},"observation_digest":"sha256:a45c9be38b8981413aa69fcd74e50f26e7d4ed56bd7c9d048f557e9a95129371","observation_id":"f3e2d616-a3f3-4103-a614-aa060b0a2e05","resolution":{"observed_at":"2026-05-10T09:08:25.512197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:ca78d0ca39b397c7ec35b219093d48b754ab36b70f1a918bca0d081878a2bfd9","observation_id":"d5b2a56b-da7c-4d7e-99ac-83fb41c9db39","resolution":{"observed_at":"2026-05-10T12:20:23.136237Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.20945","last_updated":"2026-04-22T16:51:49Z","snapshot_observed_at":"2026-07-06T23:07:36.996629Z","submitted_at":"2026-04-22T16:51:49Z","title":"Breaking Bad: Interpretability-Based Safety Audits of State-of-the-Art LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T00:32:00.789721Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.20945"},"observation_digest":"sha256:5859ec3a41537b260019c1c5f30e2d6d2a8baa7c5403ea6cd0fa610776557886","observation_id":"e8af5d2c-560f-4e6f-aa37-e9d5cb70633c","resolution":{"observed_at":"2026-05-11T13:46:05.501341Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2604.22154","last_updated":"2026-04-24T01:52:54Z","snapshot_observed_at":"2026-07-06T23:08:37.811548Z","submitted_at":"2026-04-24T01:52:54Z","title":"Reliable Self-Harm Risk Screening via Adaptive Multi-Agent LLM Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T12:41:53.950049Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2604.22154"},"observation_digest":"sha256:5f3fbf682a43f9326207e45f7b43479f6b08c981ca2077a77d695fd90faaf219","observation_id":"2d758f06-8adc-4c14-8c4d-2f363f1f20ad","resolution":{"observed_at":"2026-05-11T19:06:09.741391Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-02T13:54:55.570407Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:11d2138ab9b81551d2da4f16539bf05549f24ea0972f14db495ecfd943c25c19","observation_id":"01f579b9-361c-4712-a23f-9db28b5a9661","resolution":{"observed_at":"2026-05-12T05:31:23.946872Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2605.23598","last_updated":"2026-05-22T13:06:46Z","snapshot_observed_at":"2026-08-05T16:03:21.062446Z","submitted_at":"2026-05-22T13:06:46Z","title":"When Youth Enter the Algorithmic Wild: Discovering and Understanding Potentially Harmful Teen Videos on Douyin and Kwai","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-25T04:20:27.456558Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2605.23598"},"observation_digest":"sha256:091a14e9603148f7e5294c44b539921f4991111ec10a49f73ac5a737c1bfa1e2","observation_id":"700c0b50-026d-4736-b3d9-31e4fa83de6f","resolution":{"observed_at":"2026-05-25T04:25:19.788374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2606.07335","last_updated":"2026-06-05T14:49:26Z","snapshot_observed_at":"2026-07-06T23:47:00.425715Z","submitted_at":"2026-06-05T14:49:26Z","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T21:55:48.561400Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2606.07335"},"observation_digest":"sha256:0236c8e24c0084667a132ea36f8217e7c5735c516adb95193f7249abd24d4194","observation_id":"c61f411c-1406-4b77-8914-6f10225690b0","resolution":{"observed_at":"2026-07-02T17:37:14.897253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":"2310.17389","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-07-03T14:48:32.654187Z","title":"Toxic- chat: Unveiling hidden challenges of toxicity detec- tion in real-world user-ai conversation","venue":null,"work_id":"934340c1-de04-43a3-af48-deb728f15d44","year":2023},"citing_paper":{"arxiv_id":"2607.02079","last_updated":"2026-07-02T12:21:16Z","snapshot_observed_at":"2026-08-06T21:17:29.886069Z","submitted_at":"2026-07-02T12:21:16Z","title":"HaloGuard 1.0: An Open Weights Constitutional Classifier for Multilingual AI Safety","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-03T14:38:55.045628Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2607.02079"},"observation_digest":"sha256:57dffce10078eaa436e5c804ea4767e83eb906cde35782bb5f3db3dd28f6f3f6","observation_id":"59558a22-481c-44f1-a762-506f6272a304","resolution":{"observed_at":"2026-07-03T14:48:32.655714Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-02T14:52:12.591220Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22545","last_updated":"2026-05-06T10:14:30Z","snapshot_observed_at":"2026-08-06T13:04:33.975594Z","submitted_at":"2026-05-06T10:14:30Z","title":"Semalith v1.4: A Calibrated 184M Safety Classifier Achieving State-of-the-Art Prompt-Injection Detection at 44x Fewer Parameters than Llama-Guard-3-8B","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T14:52:12.591220Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2607.22545"},"observation_digest":"sha256:b081539fea95eb6b8235b0b76c23ac6bb30e781b021d2b94a5e7f0feb3615c36","observation_id":"048e4edc-5e2d-46cd-aa71-38c0bc50b6ab","resolution":{"observed_at":"2026-08-02T14:52:12.591220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17389","snapshot_observed_at":"2026-08-04T01:07:16.653669Z","title":"ToxicChat: Unveiling hidden challenges of toxicity detection in real-world user-AI conversation.arXiv preprint arXiv:2310.17389,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00180","last_updated":"2026-08-04T02:59:21Z","snapshot_observed_at":"2026-08-07T23:09:41.591880Z","submitted_at":"2026-07-31T18:05:25Z","title":"A Constitution-Grid Instrument for Data-Efficient RL Alignment (C-Guard)","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T01:07:16.653669Z"},"links":{"cited_paper":"/paper/2310.17389","citing_paper":"/paper/2608.00180"},"observation_digest":"sha256:351a3ca9a1eadb6060d38da5bd48e828bf44fcc7ef1f71cecba5c54e4de799f1","observation_id":"2fd0eeb0-ba99-451f-a1ed-ebdbb1a87b24","resolution":{"observed_at":"2026-08-04T01:07:16.653669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.17389/citation-record","integrity":"/paper/2310.17389/integrity","json":"/paper/2310.17389/citation-record.json","paper":"/paper/2310.17389"},"outbound":[],"paper":{"arxiv_id":"2310.17389","last_updated":"2023-10-26T13:35:41Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:38:56.540232Z","submitted_at":"2023-10-26T13:35:41Z","title":"ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2310.17389."}