{"as_of":"2026-08-08T22:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:afe94780bc67ac1a0347584082377b65b0f82f9efbeebb9e76e39deb54b2bafb","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":39,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:41:10.294985Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":16,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:6ebdb06240dbb9c4a17ccf8db1fe40e8570231c836d13f380e02971151917440","observation_id":"cde53581-e803-4c34-a549-6c02d378077d","resolution":{"observed_at":"2026-05-15T02:20:44.556311Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2408.12935","last_updated":"2026-05-13T07:56:42Z","snapshot_observed_at":"2026-08-02T12:48:59.218457Z","submitted_at":"2024-08-23T09:33:48Z","title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-23T21:54:26.670284Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2408.12935"},"observation_digest":"sha256:ede8cf8eccc58d30db1f3f87d959ed828b476d37174539fb81f8627f2f081fb6","observation_id":"62c434ea-a5db-40fe-9784-2908dbadbd76","resolution":{"observed_at":"2026-05-23T21:55:50.409899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2503.02574","last_updated":"2026-05-18T17:54:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-04T12:55:07Z","title":"LLM-Safety Evaluations Lack Robustness","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T01:26:45.402983Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2503.02574"},"observation_digest":"sha256:7c60243a97c463de7f578ad668aa363ae1c993f1568f02311a865281d36fab02","observation_id":"dae4e495-d378-4606-8300-f0e20fe5a30b","resolution":{"observed_at":"2026-05-23T01:27:21.401052Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2505.14226","last_updated":"2026-04-07T12:14:38Z","snapshot_observed_at":"2026-08-08T04:00:07.253604Z","submitted_at":"2025-05-20T11:35:25Z","title":"Phonetic Perturbations Reveal Tokenizer-Rooted Safety Gaps in LLMs","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T14:40:58.506345Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.14226"},"observation_digest":"sha256:26e8ed34c51dfa7acd1e00432546b2ed579ab0a7bcd9f570fedc96e63143e0d4","observation_id":"6c695795-7c4c-4210-9327-852985dffe04","resolution":{"observed_at":"2026-05-22T14:41:41.440784Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T15:41:10.294985Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14316","last_updated":"2025-05-20T13:03:15Z","snapshot_observed_at":"2026-08-08T01:19:21.267226Z","submitted_at":"2025-05-20T13:03:15Z","title":"Exploring Jailbreak Attacks on LLMs through Intent Concealment and Diversion","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T15:41:10.294985Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.14316"},"observation_digest":"sha256:870729633b5ddbb382614434166eb97db13d526d045a43563028ad87b24ca517","observation_id":"e7cfc789-826b-4fe4-8f1e-0e0814606f5e","resolution":{"observed_at":"2026-08-07T15:41:10.294985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T15:09:57.250696Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17147","last_updated":"2025-05-22T08:22:57Z","snapshot_observed_at":"2026-08-08T03:05:46.469319Z","submitted_at":"2025-05-22T08:22:57Z","title":"MTSA: Multi-turn Safety Alignment for LLMs through Multi-round Red-teaming","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T15:09:57.250696Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.17147"},"observation_digest":"sha256:c96661447488c844feb28ba9c112386a5f7a6ebcb743eb2580c49d027fe11477","observation_id":"634fa79b-81ca-4dbe-9899-d1b732efd882","resolution":{"observed_at":"2026-08-07T15:09:57.250696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:30:08.770366Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18807","last_updated":"2025-05-24T17:41:47Z","snapshot_observed_at":"2026-08-08T01:20:21.986624Z","submitted_at":"2025-05-24T17:41:47Z","title":"Mitigating Deceptive Alignment via Self-Monitoring","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:08.770366Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.18807"},"observation_digest":"sha256:dbe4f11cef58e1da8fdf31535adde2abcc1d9e89cc6b647bcafa4162c4679c15","observation_id":"46f8839c-2343-4faa-a81c-bff59ebf220b","resolution":{"observed_at":"2026-08-07T14:30:08.770366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:27:03.407254Z","title":"Red-teaming large language mod- els using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18889","last_updated":"2025-08-24T03:15:13Z","snapshot_observed_at":"2026-08-08T20:47:09.765405Z","submitted_at":"2025-05-24T22:22:43Z","title":"Security Concerns for Large Language Models: A Survey","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:27:03.407254Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.18889"},"observation_digest":"sha256:41c3c90b737c778f3e89f0f12ced99589b42ad0647519fa352091b179e843d16","observation_id":"cf7d8278-25e5-430d-8210-761a9ce1c0b0","resolution":{"observed_at":"2026-08-07T14:27:03.407254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:11:30.658985Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19743","last_updated":"2025-08-16T11:40:47Z","snapshot_observed_at":"2026-08-08T09:59:11.314701Z","submitted_at":"2025-05-26T09:24:36Z","title":"Token-level Accept or Reject: A Micro Alignment Approach for Large Language Models","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:30.658985Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.19743"},"observation_digest":"sha256:c0efb9fe12111cd11cf664ac24473f7fc014c9ab3905a41db4e1058bc50ec958","observation_id":"1d45dbd4-4bca-4e1b-ba73-570d344eedba","resolution":{"observed_at":"2026-08-07T14:11:30.658985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T14:15:22.576261Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20359","last_updated":"2025-05-29T13:19:08Z","snapshot_observed_at":"2026-08-07T14:07:21.986904Z","submitted_at":"2025-05-26T08:01:37Z","title":"Risk-aware Direct Preference Optimization under Nested Risk Measure","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:22.576261Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.20359"},"observation_digest":"sha256:82829f82df89c3dcd83292d48a879d04630fdc4e25b34baf4ab90261389c2c16","observation_id":"be60ffa7-ae5f-4129-a599-cc034230db8d","resolution":{"observed_at":"2026-08-07T14:15:22.576261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T12:42:11.476447Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24019","last_updated":"2025-05-29T21:39:08Z","snapshot_observed_at":"2026-08-07T20:58:14.169872Z","submitted_at":"2025-05-29T21:39:08Z","title":"LLM Agents Should Employ Security Principles","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:42:11.476447Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.24019"},"observation_digest":"sha256:66e4bbf5e0e77dea6ddfc606c53c64554768677ea5ed9af1d3b36f7f2d498747","observation_id":"dbcb2b21-ad46-4d62-a163-76dcfe37b233","resolution":{"observed_at":"2026-08-07T12:42:11.476447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T12:35:20.556699Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.24369","last_updated":"2025-05-30T09:02:07Z","snapshot_observed_at":"2026-08-08T15:50:25.845470Z","submitted_at":"2025-05-30T09:02:07Z","title":"Adversarial Preference Learning for Robust LLM Alignment","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:20.556699Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2505.24369"},"observation_digest":"sha256:1c174035795fc61a58bf2be25662aa1437bff984ba809294ce5e8dfcf2baec71","observation_id":"78c1836b-0f38-4111-aa6c-981b6d6ae984","resolution":{"observed_at":"2026-08-07T12:35:20.556699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T05:35:09.998719Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10022","last_updated":"2025-06-09T12:02:39Z","snapshot_observed_at":"2026-08-07T05:25:53.532490Z","submitted_at":"2025-06-09T12:02:39Z","title":"LLMs Caught in the Crossfire: Malware Requests and Jailbreak Challenges","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T05:35:09.998719Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.10022"},"observation_digest":"sha256:4c1c17c6ec6a9d9060f8fa462834cd77f668fabba2814e82a84adf9bcd1fd145","observation_id":"ece555ab-f3a1-4a6a-89d4-87c87c5497b5","resolution":{"observed_at":"2026-08-07T05:35:09.998719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T10:17:26.657917Z","title":"Red-teaming large language mod- els using chain of utterances for safety-alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.657917Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:89005ac513d783a8ca70abc6047cb465c24c2e3a7272f6ed450fd83b88771206","observation_id":"73959c27-4331-4ad2-ac15-f280e44a1f53","resolution":{"observed_at":"2026-08-07T10:17:26.657917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-07T01:06:08.903210Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12148","last_updated":"2025-06-13T18:08:19Z","snapshot_observed_at":"2026-08-07T00:54:57.748378Z","submitted_at":"2025-06-13T18:08:19Z","title":"Hatevolution: What Static Benchmarks Don't Tell Us","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T01:06:08.903210Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.12148"},"observation_digest":"sha256:2cafc34eb6f707f11727a87f02b6423983792f2447e279fe2788865fdc26002a","observation_id":"c2c388da-08b8-437f-9576-e73b9fee28e6","resolution":{"observed_at":"2026-08-07T01:06:08.903210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T23:49:52.753276Z","title":"arXiv preprint arXiv:2308.09662","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.16322","last_updated":"2025-06-19T13:56:41Z","snapshot_observed_at":"2026-08-07T20:46:18.050527Z","submitted_at":"2025-06-19T13:56:41Z","title":"PL-Guard: Benchmarking Language Model Safety for Polish","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T23:49:52.753276Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2506.16322"},"observation_digest":"sha256:6af3ba2901e249b3d49f8b4e34fda54a3c18f46893fab82d0e006d9ab5c072b4","observation_id":"2e1182d2-c42d-48c9-8d81-b7028e4a16c1","resolution":{"observed_at":"2026-08-06T23:49:52.753276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T20:45:53.245262Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02057","last_updated":"2025-07-02T18:00:49Z","snapshot_observed_at":"2026-08-06T20:36:51.169138Z","submitted_at":"2025-07-02T18:00:49Z","title":"MGC: A Compiler Framework Exploiting Compositional Blindness in Aligned LLMs for Malware Generation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:45:53.245262Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.02057"},"observation_digest":"sha256:2d23aecf92132c865183bf9fe434e7b1a2ac3dee9c61e302d5d78dc7498cba13","observation_id":"3fe0b9fa-78e1-43e1-ac5d-954235d63b18","resolution":{"observed_at":"2026-08-06T20:45:53.245262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T19:18:42.524938Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.06043","last_updated":"2025-08-06T05:32:54Z","snapshot_observed_at":"2026-08-08T05:18:06.985655Z","submitted_at":"2025-07-08T14:45:21Z","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T19:18:42.524938Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.06043"},"observation_digest":"sha256:55333e6c67b337bb82d38c4bfffaae197c17260240b6ba5058f23063a23e4633","observation_id":"dca1a89b-0b75-4bdf-9f69-555eb9834939","resolution":{"observed_at":"2026-08-06T19:18:42.524938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T17:17:27.596750Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11316","last_updated":"2025-07-15T13:48:35Z","snapshot_observed_at":"2026-08-08T00:39:39.784862Z","submitted_at":"2025-07-15T13:48:35Z","title":"Internal Value Alignment in Large Language Models through Controlled Value Vector Activation","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T17:17:27.596750Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.11316"},"observation_digest":"sha256:6a3bd15d1c8237dc7530430525c4e175054e35d3b74a837cd8143ce69cd1bc78","observation_id":"48420487-1b02-4ebc-8055-a103043207ed","resolution":{"observed_at":"2026-08-06T17:17:27.596750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T17:16:33.404516Z","title":"Red-teaming large language models using chain of utterances for safety-alignment, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11344","last_updated":"2025-07-15T14:20:23Z","snapshot_observed_at":"2026-08-08T00:39:15.872422Z","submitted_at":"2025-07-15T14:20:23Z","title":"Guiding LLM Decision-Making with Fairness Reward Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T17:16:33.404516Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2507.11344"},"observation_digest":"sha256:d7369c1b0c20652abf4e82a8582238af93361327e3671be6dcb7df217f60b57a","observation_id":"dc893af8-465b-4303-8687-e6e0658b8567","resolution":{"observed_at":"2026-08-06T17:16:33.404516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T17:26:48.677262Z","title":"Red-teaming large language models using chain of utterances for safety-alignment.arXiv preprint arXiv:2308.09662, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.16318","last_updated":"2025-09-01T08:35:27Z","snapshot_observed_at":"2026-08-07T01:49:06.395378Z","submitted_at":"2025-08-22T11:57:55Z","title":"SATORI: Static Test Oracle Generation for REST APIs","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-05T17:26:48.677262Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2508.16318"},"observation_digest":"sha256:7dbe4e9ffe680cba45e425c10bd54b53ba4024c24fac442267f7beedf76d5609","observation_id":"e31d8f53-d82f-4ae8-8d80-108c0eabed2d","resolution":{"observed_at":"2026-08-05T17:26:48.677262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2509.10546","last_updated":"2026-04-24T18:29:01Z","snapshot_observed_at":"2026-08-02T04:58:06.974646Z","submitted_at":"2025-09-07T22:35:15Z","title":"Learning to Conceal Risk: Controllable Multi-turn Red Teaming for LLMs in the Financial Domain","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-18T17:49:42.112564Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2509.10546"},"observation_digest":"sha256:548f17c8194333587ec99d2fd4cca68bdec41db93cff68619ff8ebac12897132","observation_id":"4b6e49a8-3215-48d8-8cea-84d2b7846095","resolution":{"observed_at":"2026-05-18T17:51:41.967482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2509.11206","last_updated":"2026-04-20T05:43:30Z","snapshot_observed_at":"2026-08-07T05:35:56.514953Z","submitted_at":"2025-09-14T10:24:13Z","title":"Evalet: Evaluating Large Language Models through Functional Fragmentation","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T16:57:25.259866Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2509.11206"},"observation_digest":"sha256:fd0078396b71af5d2f6924effe679f646aa68811a2555afe9b9a895fb71eb9e6","observation_id":"bae77025-96af-4599-a740-cfa343fa042e","resolution":{"observed_at":"2026-05-18T17:01:40.065888Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2510.09689","last_updated":"2026-04-17T02:42:09Z","snapshot_observed_at":"2026-07-06T22:32:22.953630Z","submitted_at":"2025-10-09T09:44:14Z","title":"When Search Goes Wrong: Red-Teaming Web-Augmented Large Language Models","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T09:29:14.842228Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2510.09689"},"observation_digest":"sha256:78c2055687710721c81fe432d1f14f36d78f49991315baae3a2e7ac654d34b5b","observation_id":"d4b5a365-d421-46b2-b351-7573e17cf87c","resolution":{"observed_at":"2026-05-18T09:31:11.600666Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-04T09:25:36.898264Z","title":"Red-teaming large language models using chain of utterances for safety-alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.15476","last_updated":"2026-07-04T04:20:15Z","snapshot_observed_at":"2026-08-06T02:45:04.316908Z","submitted_at":"2025-10-17T09:38:54Z","title":"SoK: Systematizing LLM Prompt Security: Taxonomies, Datasets, and Unified Evaluation of Attacks and Defenses","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T09:25:36.898264Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2510.15476"},"observation_digest":"sha256:e86f13617f7dc744e5c74a144b8bbf7be9d9e4c8ff7650ad06bcec5f58db532b","observation_id":"e41f6a8a-e9f9-4042-9641-5bcc6b2df7a1","resolution":{"observed_at":"2026-08-04T09:25:36.898264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-03T22:03:17.979326Z","title":"ArXivabs/2308.09662(2023),https://api","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.12487","last_updated":"2026-01-24T06:07:58Z","snapshot_observed_at":"2026-08-06T03:06:07.028822Z","submitted_at":"2025-11-16T07:47:31Z","title":"ToxSearch: Evolving Prompts for Toxicity Search in Large Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T22:03:17.979326Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2511.12487"},"observation_digest":"sha256:dccd720295ce3aa18e01853355777199e2b737dda4e0407fa07960a62eaf692b","observation_id":"3c6a59d3-cb9a-4631-be2b-e88e2783613e","resolution":{"observed_at":"2026-08-03T22:03:17.979326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2601.17887","last_updated":"2026-05-17T15:59:35Z","snapshot_observed_at":"2026-08-02T19:18:47.080046Z","submitted_at":"2026-01-25T15:42:01Z","title":"When Personalization Legitimizes Risks: Uncovering Safety Vulnerabilities in Personalized Dialogue Agents","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T15:11:04.394636Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2601.17887"},"observation_digest":"sha256:094ce8abb52cf718851e7297ad7c5262eb9c1885f257bb79423c1607543f2f67","observation_id":"7e1d4702-9bb7-43e7-bb08-5e519c59d389","resolution":{"observed_at":"2026-05-21T15:14:13.377473Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2601.20981","last_updated":"2026-04-21T09:20:29Z","snapshot_observed_at":"2026-08-03T04:22:47.507604Z","submitted_at":"2026-01-28T19:29:54Z","title":"Diversifying Toxicity Search in Large Language Models Through Speciation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T09:40:40.955703Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2601.20981"},"observation_digest":"sha256:856346888cb527249d7725859d89068fa0bf030fe3f69f80e6f0cca7b3208793","observation_id":"ea431d66-5646-44b9-8f16-ae81d0743128","resolution":{"observed_at":"2026-05-16T09:40:48.715471Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.01687","last_updated":"2026-05-03T02:55:30Z","snapshot_observed_at":"2026-07-06T23:14:52.417213Z","submitted_at":"2026-05-03T02:55:30Z","title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T16:00:32.413225Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.01687"},"observation_digest":"sha256:1d2e61f2de1c7f0cb3d3975187c7f767b67d5a7e97c5a067620e0ad95db87cce","observation_id":"8e386a28-aaa2-4d51-9187-1f729c03705b","resolution":{"observed_at":"2026-05-11T09:31:01.284910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.04446","last_updated":"2026-05-06T03:21:38Z","snapshot_observed_at":"2026-07-06T23:17:13.770090Z","submitted_at":"2026-05-06T03:21:38Z","title":"Misrouter: Exploiting Routing Mechanisms for Input-Only Attacks on Mixture-of-Experts LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T18:07:00.247161Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.04446"},"observation_digest":"sha256:4f59fc8e1855d5c4ba97fd90e9afbac39fb853eb2de58afa10a1d5b433a6ee77","observation_id":"65f6e4cf-e496-4a14-98f8-2f9057019a7c","resolution":{"observed_at":"2026-05-09T06:45:44.444934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.04992","last_updated":"2026-05-06T14:52:22Z","snapshot_observed_at":"2026-07-06T23:17:43.402046Z","submitted_at":"2026-05-06T14:52:22Z","title":"You Snooze, You Lose: Automatic Safety Alignment Restoration through Neural Weight Translation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-08T17:02:20.836208Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.04992"},"observation_digest":"sha256:83b58cd7a86eaa63de3da6d8c8eca2bfcf860fee1004ca120da07ae54d066015","observation_id":"c7778730-89ca-470d-b42a-0ffd009ed8b2","resolution":{"observed_at":"2026-05-11T17:51:08.429806Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2605.19190","last_updated":"2026-05-18T23:34:53Z","snapshot_observed_at":"2026-08-02T02:36:37.244109Z","submitted_at":"2026-05-18T23:34:53Z","title":"Going PLACES: Participatory Localized Red Teaming for Text-to-Image Safety in the Global South","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T07:06:57.555070Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2605.19190"},"observation_digest":"sha256:7d06fa306383b9e57369e52ccb24071f196fdfecb80b32eca9e95caf644c7faf","observation_id":"6d3603b4-cad3-4497-ae07-151bd157bf72","resolution":{"observed_at":"2026-05-20T07:08:07.009492Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.00686","last_updated":"2026-05-30T11:49:42Z","snapshot_observed_at":"2026-08-04T22:47:09.122420Z","submitted_at":"2026-05-30T11:49:42Z","title":"Dialectics of Alignment: Harnessing Unsafe Knowledge for Dynamic Safety Routing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T19:23:10.275977Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.00686"},"observation_digest":"sha256:b4809c3ebc2e64bbd9515f0f311b656ae0d1aa81394faf5da8e486048db877f5","observation_id":"6991985d-2de9-49f6-bd3c-a95fd28a1d48","resolution":{"observed_at":"2026-06-28T19:32:34.566985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.02530","last_updated":"2026-06-01T17:38:12Z","snapshot_observed_at":"2026-08-05T11:28:53.105574Z","submitted_at":"2026-06-01T17:38:12Z","title":"SafeSteer: Localized On-Policy Distillation for Efficient Safety Alignment","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-06-28T14:39:11.178976Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.02530"},"observation_digest":"sha256:6c613d2271b289d8e12063b67c2a1f7906e838d2c80b59da5bd9fc62ddc23631","observation_id":"25061803-f922-4013-9920-04f853853a18","resolution":{"observed_at":"2026-07-01T23:06:20.835743Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.07335","last_updated":"2026-06-05T14:49:26Z","snapshot_observed_at":"2026-07-06T23:47:00.425715Z","submitted_at":"2026-06-05T14:49:26Z","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-06-27T21:55:48.561400Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.07335"},"observation_digest":"sha256:1486287c175ab9b166791b7b8a4b58c4c0d0cf9a5c8dabeb4e9ea135ffcf1282","observation_id":"2dba2dfc-6332-40b8-9f6a-2f3f983906c9","resolution":{"observed_at":"2026-07-02T17:37:14.894469Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":"2308.09662","doi":"10.48550/arxiv.2308.09662","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"and Poria, S","venue":"arXiv (Cornell University)","work_id":"829e5d9e-ef8c-4b87-84c4-1f759d22c9e4","year":2023},"citing_paper":{"arxiv_id":"2606.24166","last_updated":"2026-06-23T05:40:05Z","snapshot_observed_at":"2026-07-06T23:58:49.574536Z","submitted_at":"2026-06-23T05:40:05Z","title":"Distributed Quality-Diversity Search for Toxicity in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-25T22:12:27.301100Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2606.24166"},"observation_digest":"sha256:5100bb5f0e8166e5b43a082b5a5a0fe54413dfd22b4f7f34cc42b3b9bbeb3eb4","observation_id":"587eacad-b34c-4eed-ae8e-424a3e580995","resolution":{"observed_at":"2026-07-04T19:00:04.582802Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-04T01:16:12.719886Z","title":"Red-teaming large language mod- els using chain of utterances for safety-alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00134","last_updated":"2026-07-31T14:04:49Z","snapshot_observed_at":"2026-08-07T20:59:09.642310Z","submitted_at":"2026-07-31T14:04:49Z","title":"Stateful Cooperative Agents Safeguarding LLMs Against Evolving Multi-Turn Attacks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T01:16:12.719886Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2608.00134"},"observation_digest":"sha256:0f3cfd337fb91123ef33d587370e7627a51739f15d82aad1b98941c193fee896","observation_id":"5aa005cc-1aac-45f5-b27f-407bd964ef24","resolution":{"observed_at":"2026-08-04T01:16:12.719886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T00:18:48.516596Z","title":"Red-teaming large language models using chain of utterances for safety-alignment.arXiv preprint arXiv:2308.09662, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01414","last_updated":"2026-08-02T17:49:07Z","snapshot_observed_at":"2026-08-08T15:17:45.503947Z","submitted_at":"2026-08-02T17:49:07Z","title":"No Single Neuron of Failure: Distributed Safety Alignment Against White-Box Attacks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T00:18:48.516596Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2608.01414"},"observation_digest":"sha256:bdf93899cdc3b49d031cca96936d93b44b61df93326d37b27517b214ef297e16","observation_id":"a8e4ed37-5070-4455-b867-09c47f93ea5b","resolution":{"observed_at":"2026-08-06T00:18:48.516596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09662","snapshot_observed_at":"2026-08-06T00:29:56.632621Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.03166","last_updated":"2026-08-04T05:54:32Z","snapshot_observed_at":"2026-08-08T00:04:44.793928Z","submitted_at":"2026-08-04T05:54:32Z","title":"Adversarial Stress Testing of Role-Playing Language Agents using Multi-Agent Evaluation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T00:29:56.632621Z"},"links":{"cited_paper":"/paper/2308.09662","citing_paper":"/paper/2608.03166"},"observation_digest":"sha256:bf6b8c555feb580f494efb48991af27accc93585e45b3d4d102d95567a778b09","observation_id":"51ee87da-f0b5-42b3-9169-f8cfd2021453","resolution":{"observed_at":"2026-08-06T00:29:56.632621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.09662/citation-record","integrity":"/paper/2308.09662/integrity","json":"/paper/2308.09662/citation-record.json","paper":"/paper/2308.09662"},"outbound":[],"paper":{"arxiv_id":"2308.09662","last_updated":"2023-08-30T10:21:00Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T17:01:00.147032Z","submitted_at":"2023-08-18T16:27:04Z","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 39 inbound Pith citation observations for arXiv:2308.09662."}