{"as_of":"2026-08-09T22:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:34b987b61de9d5c105dc7212296478cf4a394590d63ebfe2fa7fbbe61c8cffd2","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":38,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T18:31:14.103240Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T21:58:59.338870Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2308.03825","last_updated":"2024-05-15T12:06:31Z","snapshot_observed_at":"2026-07-06T16:03:34.432602Z","submitted_at":"2023-08-07T16:55:20Z","title":"\"Do Anything Now\": Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-17T08:39:28.047394Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2308.03825"},"observation_digest":"sha256:e2547c54d91fae639c4da77f0ae0e5eab732b04850f8507e986b2da35ffdf22c","observation_id":"7c3f6893-7268-4a2e-a523-6cae8af40f29","resolution":{"observed_at":"2026-05-17T08:39:28.183330Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2402.10260","last_updated":"2024-08-27T03:32:47Z","snapshot_observed_at":"2026-08-04T21:32:35.483431Z","submitted_at":"2024-02-15T18:58:09Z","title":"A StrongREJECT for Empty Jailbreaks","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T21:28:02.745230Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2402.10260"},"observation_digest":"sha256:6fbd0dc901667701fb7ab36d5189a66473fd8f1711fd595c6d2d1b9d0948cb66","observation_id":"a027ddf2-6f46-496d-8edc-38e7890ff0dd","resolution":{"observed_at":"2026-05-16T21:28:02.888797Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2404.01318","last_updated":"2024-10-31T22:26:40Z","snapshot_observed_at":"2026-08-02T14:59:12.115203Z","submitted_at":"2024-03-28T02:44:02Z","title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","version":5},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-15T06:08:05.386345Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2404.01318"},"observation_digest":"sha256:d62d3e9357b48e50844b690a10f4a18abb8c333ffd404cbda4a560abe3112794","observation_id":"68191e5e-ce2a-46cf-8c6a-d91f943006b8","resolution":{"observed_at":"2026-05-15T06:08:05.541412Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:cdf6fa7c1176dfa2a8db3000127d4c7a2b912645c3c79bc9ffd5aca873ad2f5f","observation_id":"360fcd46-7fe7-4367-866f-33c3e5b1f10f","resolution":{"observed_at":"2026-05-15T02:20:44.463792Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-09T18:31:14.103240Z","title":"How johnny can persuade llms to jailbreak them: Rethink- ing persuasion to challenge ai safety by humanizing llms.arXiv preprint arXiv:2401.06373, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00580","last_updated":"2025-02-01T22:26:30Z","snapshot_observed_at":"2026-08-09T18:23:58.529851Z","submitted_at":"2025-02-01T22:26:30Z","title":"Defense Against the Dark Prompts: Mitigating Best-of-N Jailbreaking with Prompt Evaluation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T18:31:14.103240Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2502.00580"},"observation_digest":"sha256:da26928215574fa006143dff4aac11780aba46c8abd0c4b94268050c24157372","observation_id":"208c5995-eb14-4156-9167-17de298fe388","resolution":{"observed_at":"2026-08-09T18:31:14.103240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-09T14:49:08.857892Z","title":"1, 14 Zhang, L., Hosseini, A., Bansal, H., Kazemi, M., Ku- mar, A., and Agarwal, R","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.01633","last_updated":"2025-06-25T15:31:17Z","snapshot_observed_at":"2026-08-09T14:42:26.219814Z","submitted_at":"2025-02-03T18:59:01Z","title":"Adversarial Reasoning at Jailbreaking Time","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-09T14:49:08.857892Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2502.01633"},"observation_digest":"sha256:e02e758b826306a10ccbeea8158c0112a0e1e8c83a6be546f2654a4a70e061a6","observation_id":"3b4e3b56-076f-46b6-869b-8c5334b66cb6","resolution":{"observed_at":"2026-08-09T14:49:08.857892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-09T12:47:21.810939Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safe ty by humanizing llms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.02260","last_updated":"2026-06-02T17:48:48Z","snapshot_observed_at":"2026-08-09T12:42:57.704129Z","submitted_at":"2025-02-04T12:17:08Z","title":"Position: Adversarial ML for LLMs Is Not Making Any Progress","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-09T12:47:21.810939Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2502.02260"},"observation_digest":"sha256:87e5c2a8758871fc5adf68a9505c93156c027a7d32c890a09ada765673d3b590","observation_id":"4443f256-d3fc-40f1-b725-0b908f096e92","resolution":{"observed_at":"2026-08-09T12:47:21.810939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-08T23:50:36.388248Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04040","last_updated":"2025-05-30T09:43:42Z","snapshot_observed_at":"2026-08-08T23:43:13.905441Z","submitted_at":"2025-02-06T13:01:44Z","title":"Safety Reasoning with Guidelines","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-08T23:50:36.388248Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2502.04040"},"observation_digest":"sha256:7521a237e5ba4cec1a4d43eec6bda5f6defbe47e0f2f2b2227d1d62cbbb44fe5","observation_id":"2484294f-a2ce-4a99-a523-2d19d67acf6b","resolution":{"observed_at":"2026-08-08T23:50:36.388248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-09T04:22:00.615220Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them : Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs , January 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05223","last_updated":"2025-02-05T21:50:34Z","snapshot_observed_at":"2026-08-09T05:14:54.791831Z","submitted_at":"2025-02-05T21:50:34Z","title":"KDA: A Knowledge-Distilled Attacker for Generating Diverse Prompts to Jailbreak LLMs","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-09T04:22:00.615220Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2502.05223"},"observation_digest":"sha256:8df087f2be881667e72469e1796b0ad4e8b088a3ce103e4cfcc2adf468818e0a","observation_id":"8d14b25a-76ed-4e01-97ce-a13af257036e","resolution":{"observed_at":"2026-08-09T04:22:00.615220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-07T21:41:06.430697Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09687","last_updated":"2025-02-13T15:15:53Z","snapshot_observed_at":"2026-08-08T23:01:51.671696Z","submitted_at":"2025-02-13T15:15:53Z","title":"Mind What You Ask For: Emotional and Rational Faces of Persuasion by Large Language Models","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T21:41:06.430697Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2502.09687"},"observation_digest":"sha256:19a4a11dda6225af357926d9b7ffaaf198a852fbbf7dc9282cfafe25e745e802","observation_id":"7dca48dd-3bf9-4c77-8c86-b88418aa5525","resolution":{"observed_at":"2026-08-07T21:41:06.430697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2505.10846","last_updated":"2026-04-16T13:52:09Z","snapshot_observed_at":"2026-08-03T03:52:26.701470Z","submitted_at":"2025-05-16T04:37:12Z","title":"AutoRAN: Automated Hijacking of Safety Reasoning in Large Reasoning Models","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T15:33:04.194346Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2505.10846"},"observation_digest":"sha256:1a9d5b609d325c5bbc302a2a35e0fb85fcb6b7edbc951f348b3bd306af99d95e","observation_id":"19f96e24-f242-4c4e-9020-7f5cb9eb2ffb","resolution":{"observed_at":"2026-05-22T15:34:57.558902Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-07T13:36:00.518360Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21277","last_updated":"2025-05-28T14:16:10Z","snapshot_observed_at":"2026-08-09T03:35:23.473114Z","submitted_at":"2025-05-27T14:48:44Z","title":"Breaking the Ceiling: Exploring the Potential of Jailbreak Attacks through Expanding Strategy Space","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T13:36:00.518360Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2505.21277"},"observation_digest":"sha256:37c17dce375cb275c37622306a6ae74899b6352c1c87409f7a60ff6aaf5bf69a","observation_id":"77ebcfc6-6a88-4f9b-b1e4-354ab618fd60","resolution":{"observed_at":"2026-08-07T13:36:00.518360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-07T12:35:23.317490Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24369","last_updated":"2025-05-30T09:02:07Z","snapshot_observed_at":"2026-08-09T16:12:29.442194Z","submitted_at":"2025-05-30T09:02:07Z","title":"Adversarial Preference Learning for Robust LLM Alignment","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:23.317490Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2505.24369"},"observation_digest":"sha256:62659b790a278c60a1ed77964f4d157ce30a569e20c993b1c12f8a3194d7b701","observation_id":"f94732ee-cc52-443f-8f74-e6808cdece7c","resolution":{"observed_at":"2026-08-07T12:35:23.317490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-07T05:47:33.123729Z","title":"In33rd USENIX Security Symposium (USENIX Security 24), pages 4657–4674","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07031","last_updated":"2026-06-25T06:36:02Z","snapshot_observed_at":"2026-08-07T23:29:11.903646Z","submitted_at":"2025-06-08T07:45:48Z","title":"HauntAttack: When Attack Follows Reasoning as a Shadow","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:47:33.123729Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2506.07031"},"observation_digest":"sha256:03b702cb21458c37fdfc0ea6596349bbf57b8a646e9551a93f7d256e0aa5c336","observation_id":"60366104-391d-4310-a74b-18c7f08b0d10","resolution":{"observed_at":"2026-08-07T05:47:33.123729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-07T01:02:31.370445Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12274","last_updated":"2025-06-13T23:03:11Z","snapshot_observed_at":"2026-08-09T00:29:06.890225Z","submitted_at":"2025-06-13T23:03:11Z","title":"InfoFlood: Jailbreaking Large Language Models with Information Overload","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T01:02:31.370445Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2506.12274"},"observation_digest":"sha256:a1090258b122e5ed4f60c45a661859576918a97156b98ce9e68e35b4a2efdab6","observation_id":"9b10db3e-a698-4987-a2eb-a75b8bdc3bc4","resolution":{"observed_at":"2026-08-07T01:02:31.370445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-06T20:45:53.371585Z","title":"arXiv preprint arXiv:2401.06373 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02057","last_updated":"2025-07-02T18:00:49Z","snapshot_observed_at":"2026-08-09T00:37:18.838001Z","submitted_at":"2025-07-02T18:00:49Z","title":"MGC: A Compiler Framework Exploiting Compositional Blindness in Aligned LLMs for Malware Generation","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T20:45:53.371585Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2507.02057"},"observation_digest":"sha256:adba337ae773180b0032b8b556ded553692f30493a191e7006e67281f8cf7165","observation_id":"e5e45584-1a69-4502-9c71-787d370e9609","resolution":{"observed_at":"2026-08-06T20:45:53.371585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-06T17:03:17.163189Z","title":"Yi Zeng, Hongpeng Lin, Jingwen Zhang, Diyi Yang, Ruoxi Jia, and Weiyan Shi","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11878","last_updated":"2026-07-06T01:46:44Z","snapshot_observed_at":"2026-08-06T16:57:07.977935Z","submitted_at":"2025-07-16T03:48:03Z","title":"LLMs Encode Harmfulness and Refusal Separately","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:03:17.163189Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2507.11878"},"observation_digest":"sha256:e5a464aa8757b398bb8dddf39f42a5288499c5b8685401d0f092f2629065a42c","observation_id":"92dd1aa6-0f85-47a0-a65a-0615be0521d5","resolution":{"observed_at":"2026-08-06T17:03:17.163189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-06T16:28:53.616114Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.13474","last_updated":"2025-07-17T18:33:50Z","snapshot_observed_at":"2026-08-09T05:14:52.776386Z","submitted_at":"2025-07-17T18:33:50Z","title":"Paper Summary Attack: Jailbreaking LLMs through LLM Safety Papers","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T16:28:53.616114Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2507.13474"},"observation_digest":"sha256:0acc7f51780f5506926d779de238418993919d8af9290a14c1c2565bc8758382","observation_id":"4fd72723-3dd6-45be-8ceb-8ac186636391","resolution":{"observed_at":"2026-08-06T16:28:53.616114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-06T14:45:20.573789Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17922","last_updated":"2025-07-23T20:39:14Z","snapshot_observed_at":"2026-08-06T14:37:37.075855Z","submitted_at":"2025-07-23T20:39:14Z","title":"From Seed to Harvest: Augmenting Human Creativity with AI for Red-teaming Text-to-Image Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.573789Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2507.17922"},"observation_digest":"sha256:2a4636c296a838630ec14d1fb591ea73b23b21cbefba76995002cf2e521af546","observation_id":"8f4ce4b1-eec0-4b31-a529-0b72a386439d","resolution":{"observed_at":"2026-08-06T14:45:20.573789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2508.04204","last_updated":"2026-05-06T06:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T08:35:10Z","title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-19T01:02:07.088724Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2508.04204"},"observation_digest":"sha256:db6eaeee51ffb9bb7af6066cfe38383642118ac1e356e68f5f9a1c4610475c4e","observation_id":"0d545991-df40-428d-8eab-c84417939a76","resolution":{"observed_at":"2026-05-19T01:02:54.829979Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-05T20:31:39.578072Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10404","last_updated":"2025-08-14T07:12:44Z","snapshot_observed_at":"2026-08-08T02:31:17.622343Z","submitted_at":"2025-08-14T07:12:44Z","title":"Layer-Wise Perturbations via Sparse Autoencoders for Adversarial Text Generation","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-05T20:31:39.578072Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2508.10404"},"observation_digest":"sha256:0563f39b6c5f4324bda895d95a304d2f8c155c352570a47253b3580ce1890830","observation_id":"a1cd6c39-faad-47d4-9419-a2595bb633ec","resolution":{"observed_at":"2026-08-05T20:31:39.578072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2508.10880","last_updated":"2026-05-08T16:28:36Z","snapshot_observed_at":"2026-07-06T22:13:04.624677Z","submitted_at":"2025-08-14T17:49:09Z","title":"Searching for Privacy Risks in LLM Agents via Simulation","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T22:40:31.407606Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2508.10880"},"observation_digest":"sha256:f5bb8d1206bc4c45a1e59370630d73cec4b02165485c4f1f61967f80c8b4bc9d","observation_id":"cb317bd1-0ce4-40ab-8e20-63a4b38816a9","resolution":{"observed_at":"2026-05-18T22:41:53.120499Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-04T10:22:09.180107Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.10271","last_updated":"2026-06-26T00:42:37Z","snapshot_observed_at":"2026-08-04T14:50:50.060411Z","submitted_at":"2025-10-11T16:14:56Z","title":"MetaBreak: Jailbreaking Online LLM Services via Special Token Manipulation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T10:22:09.180107Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2510.10271"},"observation_digest":"sha256:c21d031a28d6edeb0ad8643713eaf8c997a0e544403675416510be64e4f5b3a3","observation_id":"bee77cb5-d75b-416c-a846-777e694ab016","resolution":{"observed_at":"2026-08-04T10:22:09.180107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-04T08:04:30.692057Z","title":"Static Persuader","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.22768","last_updated":"2026-06-02T23:23:16Z","snapshot_observed_at":"2026-08-06T02:22:09.994923Z","submitted_at":"2025-10-26T17:39:21Z","title":"Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T08:04:30.692057Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2510.22768"},"observation_digest":"sha256:815b757f688f9a2a9683ce18fddfb2c0039174808d6e8957275a4206dec1f9ac","observation_id":"0642bbd3-0f80-49d4-92d0-5955c0b8acf8","resolution":{"observed_at":"2026-08-04T08:04:30.692057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2511.02356","last_updated":"2026-04-20T12:25:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-04T08:24:22Z","title":"ASTRA: An Automated Framework for Strategy Discovery, Retrieval, and Evolution for Jailbreaking LLMs","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-18T01:54:22.995178Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2511.02356"},"observation_digest":"sha256:ae3f8f7422969243f7813101bd4b982fb9050fe595fb94579a24a19f3cb13d5b","observation_id":"18e21459-18b4-4ad5-a288-fa31eba8bb12","resolution":{"observed_at":"2026-05-18T01:55:37.922855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2604.04060","last_updated":"2026-04-05T11:06:13Z","snapshot_observed_at":"2026-08-02T08:01:35.924581Z","submitted_at":"2026-04-05T11:06:13Z","title":"CoopGuard: Stateful Cooperative Agents Safeguarding LLMs Against Evolving Multi-Round Attacks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T17:06:46.917177Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2604.04060"},"observation_digest":"sha256:a7cd6d37ff793f2be87e46c29c8b8e2851ec28d5b9091e146f0ad60f824b23a1","observation_id":"a95d557b-ad7e-495a-9cd7-fbb04edb84dd","resolution":{"observed_at":"2026-05-13T17:08:00.729279Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2604.10403","last_updated":"2026-04-12T01:37:45Z","snapshot_observed_at":"2026-07-06T22:59:01.129724Z","submitted_at":"2026-04-12T01:37:45Z","title":"Latent Instruction Representation Alignment: defending against jailbreaks, backdoors and undesired knowledge in LLMs","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-10T16:41:52.440793Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2604.10403"},"observation_digest":"sha256:9880bc0950faa0cb13057f4381130aadda94d201bc5f26a2bf780f9f3a18e4ae","observation_id":"fc791e70-5c16-41dc-982d-417f1a3cede3","resolution":{"observed_at":"2026-05-11T08:21:00.078211Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2604.15780","last_updated":"2026-04-17T07:37:41Z","snapshot_observed_at":"2026-07-06T23:03:16.345488Z","submitted_at":"2026-04-17T07:37:41Z","title":"Pruning Unsafe Tickets: A Resource-Efficient Framework for Safer and More Robust LLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T09:10:55.001543Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2604.15780"},"observation_digest":"sha256:66e433886e58351582a08828cc7d992072100cb040750ab7ccf61c3e77a51851","observation_id":"516d1186-2d15-4ce0-824e-c3acbeade42e","resolution":{"observed_at":"2026-05-10T09:13:29.791947Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2605.00267","last_updated":"2026-05-04T18:25:35Z","snapshot_observed_at":"2026-08-02T06:06:01.489460Z","submitted_at":"2026-04-30T22:04:10Z","title":"Jailbroken Frontier Models Retain Their Capabilities","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-09T20:00:08.368799Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2605.00267"},"observation_digest":"sha256:68132d4cb52d003ff98b9e69b4ecc2bf72e5dbf4f80a536e175b89a2686ec32c","observation_id":"0c7e81cc-d3f9-4f7c-9653-cf251fcd6402","resolution":{"observed_at":"2026-05-11T15:26:09.694106Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2605.02647","last_updated":"2026-05-04T14:32:40Z","snapshot_observed_at":"2026-07-06T23:15:41.793878Z","submitted_at":"2026-05-04T14:32:40Z","title":"ContextualJailbreak: Evolutionary Red-Teaming via Simulated Conversational Priming","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-08T19:16:21.173184Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2605.02647"},"observation_digest":"sha256:848bf592da02f4a69c1bf3c7a31addafe547c136ada69dcdccb52b188df78cd3","observation_id":"e53c13aa-5b6c-4bc8-a939-15c5aa77f937","resolution":{"observed_at":"2026-05-09T05:55:32.403092Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-13T18:51:10.298187Z","title":"Zeng, et al., How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.00003","last_updated":"2026-03-25T15:36:15Z","snapshot_observed_at":"2026-07-13T18:51:10.077827Z","submitted_at":"2026-03-25T15:36:15Z","title":"Learning from Mistakes: Can LLM Self-Recover after Misalignment?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-13T18:51:10.298187Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2606.00003"},"observation_digest":"sha256:96196c209fb5bf90ef4a00a01b50115a8e29bb26373357427f24c93c28b495db","observation_id":"438d1711-9f9c-4492-a0d8-2e920a86953c","resolution":{"observed_at":"2026-07-13T18:51:10.298187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2606.00651","last_updated":"2026-05-30T09:54:38Z","snapshot_observed_at":"2026-07-06T23:41:19.955034Z","submitted_at":"2026-05-30T09:54:38Z","title":"MESA: Improving MoE Safety Alignment via Decentralized Expertise","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-06-28T18:52:00.377915Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2606.00651"},"observation_digest":"sha256:178f74cddf81007dc1fa64da31ca2443a454f73a25988f4c08d02bffd2f3c5bb","observation_id":"df6f768d-85fc-4e09-8fe2-c501d508f57d","resolution":{"observed_at":"2026-06-28T19:52:35.255305Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2606.05523","last_updated":"2026-06-04T00:06:13Z","snapshot_observed_at":"2026-08-08T00:57:41.523461Z","submitted_at":"2026-06-04T00:06:13Z","title":"CHASE: Adversarial Red-Blue Teaming for Improving LLM Safety using Reinforcement Learning","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-28T02:34:26.334078Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2606.05523"},"observation_digest":"sha256:3fb0ad0fdcae5b7cd4461ed968075cc22aaec82e92572286957b65b30e2c8d68","observation_id":"b48141fd-ac62-452d-bd44-8b6e0c66569e","resolution":{"observed_at":"2026-07-02T12:06:55.629602Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":"2401.06373","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-07-03T21:58:59.338870Z","title":"InProceedings of the Inter- national Conference on Learning Representations (ICLR)","venue":null,"work_id":"4dfcea58-c69c-4ded-916c-1fb984476d6d","year":2024},"citing_paper":{"arxiv_id":"2606.18193","last_updated":"2026-06-16T17:23:58Z","snapshot_observed_at":"2026-08-08T08:09:24.874844Z","submitted_at":"2026-06-16T17:23:58Z","title":"A Red-Team Study of Anthropic Fable 5 & Opus 4.8 Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T23:53:24.726702Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2606.18193"},"observation_digest":"sha256:ecce18b2024851bf14a85c518d430ac5d6ce1e8667d674d1b459c4e316b7119f","observation_id":"121405e4-9bdf-4148-addf-b8270dd4f52b","resolution":{"observed_at":"2026-07-03T21:58:59.340573Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-01T08:37:35.234400Z","title":"arXiv preprint arXiv:2401.06373","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27373","last_updated":"2026-07-29T18:25:30Z","snapshot_observed_at":"2026-08-07T20:54:40.976131Z","submitted_at":"2026-07-29T18:25:30Z","title":"RoguePrompt: Dual-Layer Encoding for Self-Reconstruction to Circumvent LLM Moderation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T08:37:35.234400Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2607.27373"},"observation_digest":"sha256:4ccac3eaa26b0b7671acbf4623d62ccccf6d0579fb27341bddece5414b50abf9","observation_id":"835104cf-0672-487a-b873-b0aa48305821","resolution":{"observed_at":"2026-08-01T08:37:35.234400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-04T01:16:10.686916Z","title":"How johnny can persuade llms to jailbreak them: Rethinking persuasion to challenge ai safety by humanizing llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00134","last_updated":"2026-07-31T14:04:49Z","snapshot_observed_at":"2026-08-07T20:59:09.642310Z","submitted_at":"2026-07-31T14:04:49Z","title":"Stateful Cooperative Agents Safeguarding LLMs Against Evolving Multi-Turn Attacks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T01:16:10.686916Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2608.00134"},"observation_digest":"sha256:c2c4c33316491f5b080cd8fdb5ded6a2de7be018d6c04edaf3dd1d3241dd2a68","observation_id":"a37eead2-fe61-4568-bb8d-e8cebb9cacac","resolution":{"observed_at":"2026-08-04T01:16:10.686916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-05T00:48:49.605652Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02665","last_updated":"2026-08-01T22:28:31Z","snapshot_observed_at":"2026-08-09T05:14:21.280717Z","submitted_at":"2026-08-01T22:28:31Z","title":"Single Canonical Prompts Underestimate LLM Safety's Surface-Form Sensitivity","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T00:48:49.605652Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2608.02665"},"observation_digest":"sha256:49ecac00c854af6f09b4a0354065e53de7df70e2e96f7a851585951a5ad14403","observation_id":"5eb38f5c-d248-4d90-a46d-cb6fd5a2b453","resolution":{"observed_at":"2026-08-05T00:48:49.605652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06373","snapshot_observed_at":"2026-08-07T00:14:50.880396Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04034","last_updated":"2026-08-03T02:28:52Z","snapshot_observed_at":"2026-08-09T14:43:30.524152Z","submitted_at":"2026-08-03T02:28:52Z","title":"A Multimodal Automatic Redteaming Evaluation based on Atomic Jailbreak Strategy Decoupling and Combination","version":1},"reference_index":157,"source":"arxiv_source","source_observed_at":"2026-08-07T00:14:50.880396Z"},"links":{"cited_paper":"/paper/2401.06373","citing_paper":"/paper/2608.04034"},"observation_digest":"sha256:2cef3ff3e53a9a4afbea24c4b7d3fbb5060808433bb1c36f85e45272d560490e","observation_id":"ddd55fd4-c3d8-43e7-af6c-f3ee3f943be8","resolution":{"observed_at":"2026-08-07T00:14:50.880396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2401.06373/citation-record","integrity":"/paper/2401.06373/integrity","json":"/paper/2401.06373/citation-record.json","paper":"/paper/2401.06373"},"outbound":[],"paper":{"arxiv_id":"2401.06373","last_updated":"2024-01-23T22:46:12Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T17:14:36.670116Z","submitted_at":"2024-01-12T16:13:24Z","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 38 inbound Pith citation observations for arXiv:2401.06373."}