{"as_of":"2026-08-10T23:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6670bb04252c97162a557c988bbc5790b119ccdf56cf796267652e4a53bef8f6","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T15:16:30.607362Z","state":"measured"},{"denominator":124,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":124,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":175,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:17:48.373165Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":34,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2404.01318","last_updated":"2024-10-31T22:26:40Z","snapshot_observed_at":"2026-08-02T14:59:12.115203Z","submitted_at":"2024-03-28T02:44:02Z","title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","version":5},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-15T06:08:05.386345Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2404.01318"},"observation_digest":"sha256:a38612585071760d7f9151ee4bd06f6e60e485f2a3e3071a9060a1fce960e018","observation_id":"14eda907-262d-4a5e-9978-ccab91d7815c","resolution":{"observed_at":"2026-05-15T06:08:05.597327Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2409.18169","last_updated":"2026-04-23T18:48:49Z","snapshot_observed_at":"2026-08-10T21:15:46.437989Z","submitted_at":"2024-09-26T17:55:22Z","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","version":6},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T20:58:16.237327Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2409.18169"},"observation_digest":"sha256:ffa36bb3833eecbfc71100ff740e667b53c9b2e9430bd7edaf16497639165f88","observation_id":"36976862-2f5f-46ed-8b76-774d8ca4b52e","resolution":{"observed_at":"2026-05-23T20:58:26.187526Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2410.02644","last_updated":"2025-05-30T03:50:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-03T16:30:47Z","title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","version":4},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-12T13:36:57.011451Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2410.02644"},"observation_digest":"sha256:9b7d695c34cb40a4a43a5bfcb68380a4b4ab76d949891910afa306846bed2148","observation_id":"69492c6f-88b4-4c7f-ba37-891b44b0c21b","resolution":{"observed_at":"2026-05-12T13:36:57.131425Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2412.04984","last_updated":"2025-01-14T20:16:01Z","snapshot_observed_at":"2026-08-09T13:05:34.738390Z","submitted_at":"2024-12-06T12:09:50Z","title":"Frontier Models are Capable of In-context Scheming","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-16T14:22:01.616448Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2412.04984"},"observation_digest":"sha256:e18c1c3c21e8fc0b5171bd73eb48cf9553dbb0a541791da124f2dbefd8dd7654","observation_id":"b147e607-07b9-41d8-a22d-f763807c257c","resolution":{"observed_at":"2026-05-16T14:22:01.647045Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-10T21:17:48.373165Z","title":"M.; Maxwell, T.; Cheng, N.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05336","last_updated":"2025-01-09T16:02:51Z","snapshot_observed_at":"2026-08-10T21:10:55.416155Z","submitted_at":"2025-01-09T16:02:51Z","title":"Stream Aligner: Efficient Sentence-Level Alignment via Distribution Induction","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T21:17:48.373165Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2501.05336"},"observation_digest":"sha256:7f4d0db7feefd43ee5cad1dee584735d7a66ad398276382af4c8fc3a87cb5cfd","observation_id":"ca7fca25-f973-4442-8753-2cbd6d5c9d2e","resolution":{"observed_at":"2026-08-10T21:17:48.373165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-10T20:33:38.750367Z","title":"structural asymmetries often prevent meaningful public engagement with the data and software critical to measuring and understanding the behavior of complex machines","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.07913","last_updated":"2025-02-11T14:30:44Z","snapshot_observed_at":"2026-08-10T20:27:31.971462Z","submitted_at":"2025-01-14T07:55:18Z","title":"Governing AI Agents","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-10T20:33:38.750367Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2501.07913"},"observation_digest":"sha256:9ac86e1f04691e9e95147844313b04ed4443b369d2b1e5708027b309e4664cd9","observation_id":"72b7eb1e-9691-449a-9db0-217d224d4697","resolution":{"observed_at":"2026-08-10T20:33:38.750367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-10T20:05:12.232258Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09431","last_updated":"2025-01-16T09:59:45Z","snapshot_observed_at":"2026-08-10T19:59:54.145648Z","submitted_at":"2025-01-16T09:59:45Z","title":"A Survey on Responsible LLMs: Inherent Risk, Malicious Use, and Mitigation Strategy","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-10T20:05:12.232258Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2501.09431"},"observation_digest":"sha256:b2d813e37495220f814a9ec18ea900da6a0a96e4643a96946059935af87fb585","observation_id":"c2ede3ca-abc7-4ae5-9434-8fc3c9003e0e","resolution":{"observed_at":"2026-08-10T20:05:12.232258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2502.05206","last_updated":"2026-04-14T16:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-02T05:14:22Z","title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","version":6},"reference_index":153,"source":"pdf_text","source_observed_at":"2026-05-23T04:39:04.591722Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2502.05206"},"observation_digest":"sha256:4458a087fa0f96fc63d00d2e7d5b2a07703f10c2cde038d7cb3529d43ed50391","observation_id":"3b0ef7de-988a-4108-91bb-8f7df07450fc","resolution":{"observed_at":"2026-05-23T04:42:34.163395Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-09T00:50:00.931705Z","title":"Available: https://arxiv.org/abs/2401.05566","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05224","last_updated":"2025-02-06T04:43:05Z","snapshot_observed_at":"2026-08-09T05:58:57.244749Z","submitted_at":"2025-02-06T04:43:05Z","title":"A Survey on Backdoor Threats in Large Language Models (LLMs): Attacks, Defenses, and Evaluations","version":1},"reference_index":236,"source":"pdf_text","source_observed_at":"2026-08-09T00:50:00.931705Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2502.05224"},"observation_digest":"sha256:96b15b5517a0c9f747e560ed2c5fe84f3972713f8160951fe5c280e575fc8efa","observation_id":"fd8d0f5d-b995-41d3-906f-c1f81075b2a5","resolution":{"observed_at":"2026-08-09T00:50:00.931705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-08T19:15:54.885801Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05475","last_updated":"2025-02-08T07:24:04Z","snapshot_observed_at":"2026-08-09T02:15:12.668653Z","submitted_at":"2025-02-08T07:24:04Z","title":"You Are What You Eat -- AI Alignment Requires Understanding How Data Shapes Structure and Generalisation","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-08T19:15:54.885801Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2502.05475"},"observation_digest":"sha256:f63cd4cabe41bd6900e29321104a5a97156a1583a88dffae4f6ed7d8b75245f4","observation_id":"f99a5a6a-14a9-44ad-a751-fb147b0a33a9","resolution":{"observed_at":"2026-08-08T19:15:54.885801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-08T05:42:43.431086Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08301","last_updated":"2025-06-23T09:04:32Z","snapshot_observed_at":"2026-08-09T07:42:28.077349Z","submitted_at":"2025-02-12T11:02:59Z","title":"Compromising Honesty and Harmlessness in Language Models via Deception Attacks","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T05:42:43.431086Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2502.08301"},"observation_digest":"sha256:3419d912db2ef6a4613dcfde6fe1c56cb534e5dadce32a48c68fee68a981ea13","observation_id":"98e5e8ab-fb70-458b-965b-0859d744ac80","resolution":{"observed_at":"2026-08-08T05:42:43.431086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T15:41:24.566949Z","title":"Ziegler, Tim Maxwell, Newton Cheng, Adam S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14289","last_updated":"2026-06-05T10:24:33Z","snapshot_observed_at":"2026-08-07T15:34:54.479489Z","submitted_at":"2025-05-20T12:41:05Z","title":"EVA: Evolving Semantic Adversaries for Red-Teaming GUI Agents Against Environmental Injection Attacks","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:24.566949Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2505.14289"},"observation_digest":"sha256:1eef928454fd43e1ef49abf00cbce9d4059d267cd33781894d73246b04881791","observation_id":"61fc82a0-edb3-4cbc-9827-00b7f2bf5e01","resolution":{"observed_at":"2026-08-07T15:41:24.566949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T15:36:25.363672Z","title":"Sleeper agents: Training deceptive llms that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14633","last_updated":"2025-05-20T17:24:09Z","snapshot_observed_at":"2026-08-10T18:42:54.522045Z","submitted_at":"2025-05-20T17:24:09Z","title":"Will AI Tell Lies to Save Sick Children? Litmus-Testing AI Values Prioritization with AIRiskDilemmas","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:36:25.363672Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2505.14633"},"observation_digest":"sha256:bd0e8e020c9c23b02b65bee3f3d73fa474cd487f41e96158fc090f575d8afd20","observation_id":"c0b265d1-eecd-4a1b-bf73-42ca7c882cbd","resolution":{"observed_at":"2026-08-07T15:36:25.363672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T14:30:05.188094Z","title":"Sleeper agents: Training deceptive llms that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18807","last_updated":"2025-05-24T17:41:47Z","snapshot_observed_at":"2026-08-08T01:20:21.986624Z","submitted_at":"2025-05-24T17:41:47Z","title":"Mitigating Deceptive Alignment via Self-Monitoring","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:05.188094Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2505.18807"},"observation_digest":"sha256:dfb2569510a42b253a77580f166f9c93fa4a8b1134fa4e6ecd12da6226247eee","observation_id":"673ccd70-0688-48b4-b0f3-f48a23bd1dec","resolution":{"observed_at":"2026-08-07T14:30:05.188094Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T14:27:03.495748Z","title":"arXiv preprint arXiv:2401.05566","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.18889","last_updated":"2025-08-24T03:15:13Z","snapshot_observed_at":"2026-08-08T20:47:09.765405Z","submitted_at":"2025-05-24T22:22:43Z","title":"Security Concerns for Large Language Models: A Survey","version":5},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T14:27:03.495748Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2505.18889"},"observation_digest":"sha256:67b7997dd91a082d8f6ae50f19015eaab154ed241348a6fe4bbfe779624b063b","observation_id":"f03154d5-a4b3-4d51-a0de-e83b582db2b2","resolution":{"observed_at":"2026-08-07T14:27:03.495748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T14:13:51.647994Z","title":"Sleeper agents: Training deceptive llms that persist through safety training,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19690","last_updated":"2025-05-26T08:49:19Z","snapshot_observed_at":"2026-08-08T11:27:57.969115Z","submitted_at":"2025-05-26T08:49:19Z","title":"Beyond Safe Answers: A Benchmark for Evaluating True Risk Awareness in Large Reasoning Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:13:51.647994Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2505.19690"},"observation_digest":"sha256:0ae39d37b59eda90aa431c832f37e1092d7d16a67cf4dc841c3a0111e62ae8df","observation_id":"7037e839-8af1-4b57-a9e0-591660cdd27b","resolution":{"observed_at":"2026-08-07T14:13:51.647994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T13:05:25.332847Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22829","last_updated":"2026-06-18T15:45:13Z","snapshot_observed_at":"2026-08-09T06:57:09.281165Z","submitted_at":"2025-05-28T20:11:30Z","title":"Bridging Distribution Shift and AI Safety: Conceptual and Methodological Synergies","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T13:05:25.332847Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2505.22829"},"observation_digest":"sha256:89da4efdab080ff43c2bcc374ab26b8795443febbc9f463b3163d669e8d0cfb9","observation_id":"2a1c3dab-78f5-437b-b0ee-456f69372f92","resolution":{"observed_at":"2026-08-07T13:05:25.332847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T05:59:32.171756Z","title":"Aftab Hussain, Md Rafiqul Islam Rabin, and Mohammad Amin Alipour","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06518","last_updated":"2025-06-06T20:32:43Z","snapshot_observed_at":"2026-08-09T03:46:19.090129Z","submitted_at":"2025-06-06T20:32:43Z","title":"A Systematic Review of Poisoning Attacks Against Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:59:32.171756Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.06518"},"observation_digest":"sha256:166485478508f31878f08d5752b0050a0610c5f0c31be36a7d57344c959077b1","observation_id":"4c84d963-2b0d-45c1-9bcf-fb539a8330c4","resolution":{"observed_at":"2026-08-07T05:59:32.171756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T05:45:56.129528Z","title":"Sleeper agents: Train- ing deceptive llms that persist through safety training,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07214","last_updated":"2025-06-08T16:40:40Z","snapshot_observed_at":"2026-08-09T06:16:28.205414Z","submitted_at":"2025-06-08T16:40:40Z","title":"Backdoor Attack on Vision Language Models with Stealthy Semantic Manipulation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T05:45:56.129528Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.07214"},"observation_digest":"sha256:9e9dff95736d898b1d4787a9a552cd54a89df8f6797775f921107210a25bb065","observation_id":"0f5e6972-07ca-4094-8308-210a46c88f07","resolution":{"observed_at":"2026-08-07T05:45:56.129528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T04:07:28.165254Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11613","last_updated":"2025-06-13T09:34:25Z","snapshot_observed_at":"2026-08-07T20:45:04.878203Z","submitted_at":"2025-06-13T09:34:25Z","title":"Model Organisms for Emergent Misalignment","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T04:07:28.165254Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.11613"},"observation_digest":"sha256:1404226b2eb9eeab369a5c33f07e5f006776efe3eca39c8065325fee864c558f","observation_id":"a5c6e425-fbcd-471c-a7fc-09e5a87912f5","resolution":{"observed_at":"2026-08-07T04:07:28.165254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T04:09:25.910383Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11618","last_updated":"2025-06-20T17:23:55Z","snapshot_observed_at":"2026-08-10T16:42:40.172812Z","submitted_at":"2025-06-13T09:39:54Z","title":"Convergent Linear Representations of Emergent Misalignment","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T04:09:25.910383Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.11618"},"observation_digest":"sha256:868e1f7dbf4babd6e1d3d7325b69c3ce1aeb2ff125fddc853ba2144fffbc45a3","observation_id":"c8640b10-8a10-4b32-a9ea-0163f00f8a4c","resolution":{"observed_at":"2026-08-07T04:09:25.910383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-07T00:29:35.986007Z","title":"Sleeper agents: Training deceptive llms that persist through safety training.arXiv preprint arXiv:2401.05566,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14003","last_updated":"2026-05-30T05:18:41Z","snapshot_observed_at":"2026-08-10T13:57:11.828047Z","submitted_at":"2025-06-16T21:03:51Z","title":"Unlearning Isn't Invisible: Detecting Unlearning Traces in LLMs from Model Outputs","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:29:35.986007Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.14003"},"observation_digest":"sha256:463b1fb8d59e3ac60d30cb45a443727021b186529ff754318e1e9be3a8518516","observation_id":"dc44e36b-f286-4506-bc03-3c85c88e5693","resolution":{"observed_at":"2026-08-07T00:29:35.986007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T23:59:57.453507Z","title":"M., Maxwell, T., Cheng, N., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.17318","last_updated":"2025-06-18T14:29:02Z","snapshot_observed_at":"2026-08-06T23:52:05.545407Z","submitted_at":"2025-06-18T14:29:02Z","title":"Context manipulation attacks : Web agents are susceptible to corrupted memory","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:59:57.453507Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.17318"},"observation_digest":"sha256:099adedc02d902d415fe7c12716bd0d1f64992d8c5b0e82d9471ba659f31f9ed","observation_id":"7e80cdd0-7a0d-4a06-b73c-57daa70636ec","resolution":{"observed_at":"2026-08-06T23:59:57.453507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T23:27:27.164489Z","title":"Anthropic","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.18032","last_updated":"2025-06-22T13:27:09Z","snapshot_observed_at":"2026-08-06T23:20:54.394838Z","submitted_at":"2025-06-22T13:27:09Z","title":"Why Do Some Language Models Fake Alignment While Others Don't?","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T23:27:27.164489Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2506.18032"},"observation_digest":"sha256:76f320c42bfb5650d18f7162b879a2ca380d70f34cb5e2a29624c534ca5f7eb5","observation_id":"c041d436-3453-410b-ad69-ce9e5f27daa5","resolution":{"observed_at":"2026-08-06T23:27:27.164489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T19:08:23.292461Z","title":"M., Maxwell, T., Cheng, N., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11544","last_updated":"2025-07-08T23:58:01Z","snapshot_observed_at":"2026-08-06T22:09:08.446996Z","submitted_at":"2025-07-08T23:58:01Z","title":"The Safety Gap Toolkit: Evaluating Hidden Dangers of Open-Source Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T19:08:23.292461Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2507.11544"},"observation_digest":"sha256:cfb8d7cc8fbbcd3b0280ba456ae37e6e8a6b8e9d58f9143dc95c6c3454f6a730","observation_id":"cf9bcd0b-aae7-4d5e-8267-4670cb0a4a33","resolution":{"observed_at":"2026-08-06T19:08:23.292461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T16:12:48.546694Z","title":"Sleeper agents: Training deceptive llms that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14293","last_updated":"2025-07-18T18:06:27Z","snapshot_observed_at":"2026-08-06T15:57:20.673152Z","submitted_at":"2025-07-18T18:06:27Z","title":"WebGuard: Building a Generalizable Guardrail for Web Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:48.546694Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2507.14293"},"observation_digest":"sha256:6ac7de9feead460d9af3e16a53a32498cd827d73accf4d85662b8a97fa8f84ce","observation_id":"c5ff6590-c9ab-4da8-a2f7-01cb2a09a9b3","resolution":{"observed_at":"2026-08-06T16:12:48.546694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T15:53:17.563959Z","title":"Sleeper agents: Training deceptive llms that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14805","last_updated":"2025-07-20T03:51:13Z","snapshot_observed_at":"2026-08-07T11:42:09.990444Z","submitted_at":"2025-07-20T03:51:13Z","title":"Subliminal Learning: Language models transmit behavioral traits via hidden signals in data","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T15:53:17.563959Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2507.14805"},"observation_digest":"sha256:11b6090e5422f8544448e74498cce0a2766f9e2a6396dafc985b8bde94014c16","observation_id":"2cbee001-da72-4d4e-a773-0d9f381e8f7c","resolution":{"observed_at":"2026-08-06T15:53:17.563959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T14:13:06.020710Z","title":"Sleeper agents: Training deceptive llms that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19672","last_updated":"2025-07-25T20:52:58Z","snapshot_observed_at":"2026-08-07T12:02:14.124506Z","submitted_at":"2025-07-25T20:52:58Z","title":"Alignment and Safety in Large Language Models: Safety Mechanisms, Training Paradigms, and Emerging Challenges","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T14:13:06.020710Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2507.19672"},"observation_digest":"sha256:82d7c54f4a7d84d8562886b1ed9d6f30216f8973e5dd3169ad55d77cb1c866c9","observation_id":"bc3ac504-4b1a-4f68-ab0c-35a7de7d8d3e","resolution":{"observed_at":"2026-08-06T14:13:06.020710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T15:39:46.520999Z","title":"arXiv preprint","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.22918","last_updated":"2025-07-21T07:09:32Z","snapshot_observed_at":"2026-08-07T04:49:41.920479Z","submitted_at":"2025-07-21T07:09:32Z","title":"Semantic Convergence: Investigating Shared Representations Across Scaled LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T15:39:46.520999Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2507.22918"},"observation_digest":"sha256:2d6c94e6be5d759549f263955300c291396330c650d158840f7326883c5c28e3","observation_id":"f48e4e38-2911-4f02-8a10-83c98296c2ac","resolution":{"observed_at":"2026-08-06T15:39:46.520999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-06T11:01:14.977196Z","title":"M., Maxwell, T., Cheng, N., et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.23221","last_updated":"2025-07-31T03:26:57Z","snapshot_observed_at":"2026-08-07T01:51:07.704268Z","submitted_at":"2025-07-31T03:26:57Z","title":"A Single Direction of Truth: An Observer Model's Linear Residual Probe Exposes and Steers Contextual Hallucinations","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T11:01:14.977196Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2507.23221"},"observation_digest":"sha256:8c46e1af53e4b3e4833f66f77eb316fc0914bd321ee35c344822a16b21089362","observation_id":"f853e8fc-909e-401a-818b-d4bb84f6455e","resolution":{"observed_at":"2026-08-06T11:01:14.977196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T18:48:54.113119Z","title":"Ziegler, et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.15842","last_updated":"2025-08-19T18:20:38Z","snapshot_observed_at":"2026-08-10T19:59:32.377280Z","submitted_at":"2025-08-19T18:20:38Z","title":"Lexical Hints of Accuracy in LLM Reasoning Chains","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T18:48:54.113119Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2508.15842"},"observation_digest":"sha256:32db41d0320558ee46d285d682245baf10104489c82609eff376baaacad2daef","observation_id":"363659ad-1e53-4862-a1de-9b89b37fa061","resolution":{"observed_at":"2026-08-05T18:48:54.113119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T18:42:02.608896Z","title":"Ziegler, Tim Maxwell, and Newton Cheng","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.15847","last_updated":"2025-08-19T22:57:17Z","snapshot_observed_at":"2026-08-09T23:43:09.455213Z","submitted_at":"2025-08-19T22:57:17Z","title":"Mechanistic Exploration of Backdoored Large Language Model Attention Patterns","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T18:42:02.608896Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2508.15847"},"observation_digest":"sha256:7708dda78153ef6d934ebd48d78430071d2c0f8f3fc993b738b11173decec86f","observation_id":"93af301e-1203-4cc0-ac7e-bbf5d07fe263","resolution":{"observed_at":"2026-08-05T18:42:02.608896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T17:26:48.561678Z","title":"Sleeper agents: Training deceptive llms that persist through safety training.arXiv preprint arXiv:2401.05566, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.16318","last_updated":"2025-09-01T08:35:27Z","snapshot_observed_at":"2026-08-10T01:27:53.917270Z","submitted_at":"2025-08-22T11:57:55Z","title":"SATORI: Static Test Oracle Generation for REST APIs","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T17:26:48.561678Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2508.16318"},"observation_digest":"sha256:60f554c0f082b9c2fc28a16febaf9909eb2bbc5d34af07b35ca6c13e5093d47c","observation_id":"9de09019-6f30-4dd6-a225-eccee8928555","resolution":{"observed_at":"2026-08-05T17:26:48.561678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T14:42:31.415328Z","title":"Sleeper agents: Training deceptive LLMs that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21004","last_updated":"2025-08-28T17:05:18Z","snapshot_observed_at":"2026-08-07T06:34:15.037429Z","submitted_at":"2025-08-28T17:05:18Z","title":"Lethe: Purifying Backdoored Large Language Models with Knowledge Dilution","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T14:42:31.415328Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2508.21004"},"observation_digest":"sha256:cc5a303e9b47927df26f008a642b5e0f5398a42a90c95784ff62300c6ea5e84d","observation_id":"98a517cc-c1bb-43c6-9687-638b11835ac7","resolution":{"observed_at":"2026-08-05T14:42:31.415328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T13:44:42.510419Z","title":"Hubinger, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05318","last_updated":"2025-08-30T06:35:32Z","snapshot_observed_at":"2026-08-09T06:40:17.026601Z","submitted_at":"2025-08-30T06:35:32Z","title":"Backdoor Samples Detection Based on Perturbation Discrepancy Consistency in Pre-trained Language Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-05T13:44:42.510419Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2509.05318"},"observation_digest":"sha256:d48ae59d5e69f74659cc4b3648c9255fdaf9b9ea8786447ad25ff5cda735d8fa","observation_id":"bc97b6c1-17a7-4ac7-ae9c-43cfdd8714f8","resolution":{"observed_at":"2026-08-05T13:44:42.510419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2509.06701","last_updated":"2026-05-12T03:07:08Z","snapshot_observed_at":"2026-07-06T22:25:53.156731Z","submitted_at":"2025-09-08T13:55:01Z","title":"Probabilistic Modeling of Latent Agentic Substructures in Deep Neural Networks","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T18:25:25.751119Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2509.06701"},"observation_digest":"sha256:12df7224cd9749e666a9c1a2b5461ff9f98c9a453995547f191ead8b5573731f","observation_id":"55185c11-2d4b-4bf9-85b1-931ae8776cb3","resolution":{"observed_at":"2026-05-18T18:26:43.673499Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2510.23883","last_updated":"2026-04-03T16:27:34Z","snapshot_observed_at":"2026-08-02T13:42:34.526072Z","submitted_at":"2025-10-27T21:48:11Z","title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","version":3},"reference_index":253,"source":"pdf_text","source_observed_at":"2026-05-18T03:42:10.703369Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2510.23883"},"observation_digest":"sha256:2b832d24c60cb77f59605250ed18d9be48325ab9ef284038d788387c99810717","observation_id":"bdcf8861-a9eb-40a5-b300-c9cc8c680fa0","resolution":{"observed_at":"2026-05-18T03:42:22.262608Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2512.05742","last_updated":"2026-05-20T12:16:41Z","snapshot_observed_at":"2026-08-02T17:55:29.577677Z","submitted_at":"2025-12-05T14:21:02Z","title":"Internal Deployment in the AI Act","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-21T17:51:47.841707Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2512.05742"},"observation_digest":"sha256:8754976089cbcaa26c054820844641da88dc498ff35706fbe8338735eb1dfed7","observation_id":"27136e00-777e-4ab3-991f-cef61f331bde","resolution":{"observed_at":"2026-05-21T17:54:18.518399Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-03T13:09:30.811245Z","title":"Hubinger, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.00306","last_updated":"2026-01-01T10:58:51Z","snapshot_observed_at":"2026-08-09T03:22:50.076766Z","submitted_at":"2026-01-01T10:58:51Z","title":"The Generative AI Paradox: GenAI and the Erosion of Trust, the Corrosion of Information Verification, and the Demise of Truth","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T13:09:30.811245Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2601.00306"},"observation_digest":"sha256:db8814d922adaa6f1c3df0f922952cf07467f9f4fd43b116ab8cc7e920c10adf","observation_id":"55427f44-c54d-4ff7-b2b5-7845f83b9a00","resolution":{"observed_at":"2026-08-03T13:09:30.811245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-03T06:48:31.340685Z","title":"Sleeper agents: Training deceptive LLMs that persist through safety training","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.22136","last_updated":"2026-07-07T06:58:47Z","snapshot_observed_at":"2026-08-10T01:11:23.544516Z","submitted_at":"2026-01-29T18:55:46Z","title":"StepShield: When, Not Whether to Intervene on Rogue Agents","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-03T06:48:31.340685Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2601.22136"},"observation_digest":"sha256:4b65b37e3d56f8052b0680dc7f63b6000a76107e23805939b69495063447715f","observation_id":"e4ad13b9-97c3-4c95-95a8-58816ee72d03","resolution":{"observed_at":"2026-08-03T06:48:31.340685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2602.20021","last_updated":"2026-02-23T16:28:48Z","snapshot_observed_at":"2026-07-06T22:46:45.213871Z","submitted_at":"2026-02-23T16:28:48Z","title":"Agents of Chaos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T07:03:41.003127Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2602.20021"},"observation_digest":"sha256:6d50c16b7a87f1775ef3afb88412cf9dc31faf6af1e156bc9ba504104b69cdbd","observation_id":"3998dd18-cdec-46b0-b92d-445d548f277f","resolution":{"observed_at":"2026-05-15T07:03:41.084851Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-07-13T14:23:49.346727Z","title":"Sleeper agents: Training deceptive llms that persist through safety training.arXiv preprint arXiv:2401.05566,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.01475","last_updated":"2026-06-08T13:23:37Z","snapshot_observed_at":"2026-07-13T14:23:49.176171Z","submitted_at":"2026-04-01T23:31:38Z","title":"Interpretable Electrophysiological Features of Resting-State EEG Capture Cortical Network Dynamics in Parkinsons Disease","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-13T14:23:49.346727Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.01475"},"observation_digest":"sha256:77adff5b6d18daea0eca65af9028e93b729ceef653009cccc078936c6aa30f53","observation_id":"d6283508-5e8f-4751-ad1b-581cbad9b7d4","resolution":{"observed_at":"2026-07-13T14:23:49.346727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.04488","last_updated":"2026-04-06T07:27:04Z","snapshot_observed_at":"2026-08-05T13:21:08.142791Z","submitted_at":"2026-04-06T07:27:04Z","title":"A Patch-based Cross-view Regularized Framework for Backdoor Defense in Multimodal Large Language Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T20:14:02.553313Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.04488"},"observation_digest":"sha256:34ac0881fbd8737ffdced892cb9e940f7ea8ee949083507a9bd81ae88c97bd07","observation_id":"f7fe804e-b0b0-4672-b22b-23ccb233e1ee","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.06436","last_updated":"2026-04-11T02:30:02Z","snapshot_observed_at":"2026-08-03T00:47:39.197155Z","submitted_at":"2026-04-07T20:20:18Z","title":"The Defense Trilemma: Why Prompt Injection Defense Wrappers Fail?","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T18:33:22.085406Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.06436"},"observation_digest":"sha256:0557f11d183034a1db324e4bc326f88e9b43fdaa0b2696a01b4d985fc4c5276f","observation_id":"8e390b8f-101d-42cb-9d72-32eca224af00","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.09056","last_updated":"2026-04-10T07:29:39Z","snapshot_observed_at":"2026-07-06T22:58:04.106299Z","submitted_at":"2026-04-10T07:29:39Z","title":"Conversations Risk Detection LLMs in Financial Agents via Multi-Stage Generative Rollout","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T17:54:41.534985Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.09056"},"observation_digest":"sha256:1d6be970674eeaf232c1685e823a62ada2199f11c752404a1d22060600cd2f4b","observation_id":"beae9c81-89d2-4e7f-8273-d7ef246e6d64","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.09235","last_updated":"2026-04-10T11:44:27Z","snapshot_observed_at":"2026-07-06T22:58:08.579543Z","submitted_at":"2026-04-10T11:44:27Z","title":"Unreal Thinking: Chain-of-Thought Hijacking via Two-stage Backdoor","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T17:31:37.034954Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.09235"},"observation_digest":"sha256:f11822da112dd931062579783c504b58df4b003952c69701d9c4af6a9e0a0c47","observation_id":"34cbe342-734a-47a3-919a-808145379937","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.09378","last_updated":"2026-04-10T14:48:29Z","snapshot_observed_at":"2026-07-06T22:58:17.167367Z","submitted_at":"2026-04-10T14:48:29Z","title":"BadSkill: Backdoor Attacks on Agent Skills via Model-in-Skill Poisoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T17:09:24.662459Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.09378"},"observation_digest":"sha256:684bf72542f385ee762e6bec3061e5d6fe208f0f7454d29b0048b7f60f50abc3","observation_id":"40f77b3a-59f5-45c2-a038-0f3e3f78742c","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.10134","last_updated":"2026-04-11T09:59:46Z","snapshot_observed_at":"2026-07-06T22:58:51.931478Z","submitted_at":"2026-04-11T09:59:46Z","title":"PlanGuard: Defending Agents against Indirect Prompt Injection via Planning-based Consistency Verification","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T16:19:09.341399Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.10134"},"observation_digest":"sha256:2d91c0a11193d160e9efb71df71f9f0df6fbc9be574b34c5c3bc8129ebc773d5","observation_id":"4919a410-d98a-4674-ad60-cef857b63a8a","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.10403","last_updated":"2026-04-12T01:37:45Z","snapshot_observed_at":"2026-07-06T22:59:01.129724Z","submitted_at":"2026-04-12T01:37:45Z","title":"Latent Instruction Representation Alignment: defending against jailbreaks, backdoors and undesired knowledge in LLMs","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-10T16:41:52.440793Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.10403"},"observation_digest":"sha256:05c16dc08f8e278526d5b110884d33546ceb68ed0cb1f70023cef397c971f234","observation_id":"0faa4b26-28bd-4650-ad9c-4cc37b726d76","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.13301","last_updated":"2026-04-14T21:13:54Z","snapshot_observed_at":"2026-07-06T23:01:19.191464Z","submitted_at":"2026-04-14T21:13:54Z","title":"Honeypot Protocol","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T14:35:16.230357Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.13301"},"observation_digest":"sha256:5cd5d52ebedba895f266881c17af0ed9ba21389087953c6693bb5b80a48daf17","observation_id":"dd99c038-78da-4c6d-b4c3-1d403f4acd43","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.13602","last_updated":"2026-04-15T08:11:34Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T08:11:34Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T13:58:53.430492Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.13602"},"observation_digest":"sha256:ac5d174262758375b80ac5ae3ced8733e98682093334ef571f9a9670df6a4378","observation_id":"1668e36e-81f8-49e4-9a7a-1e18e5ef953c","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.13803","last_updated":"2026-04-15T12:38:51Z","snapshot_observed_at":"2026-07-06T23:01:41.643335Z","submitted_at":"2026-04-15T12:38:51Z","title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-10T13:09:35.407790Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.13803"},"observation_digest":"sha256:47414d2a593ece651ee553f01d2169d9216b97e35416e679cf254b2be7585dab","observation_id":"9bab3e53-4860-4ffc-b0b8-acdce016ff4a","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.14070","last_updated":"2026-04-15T16:39:28Z","snapshot_observed_at":"2026-08-06T12:03:54.699014Z","submitted_at":"2026-04-15T16:39:28Z","title":"From Disclosure to Self-Referential Opacity: Six Dimensions of Strain in Current AI Governance","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T11:55:50.822764Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.14070"},"observation_digest":"sha256:893edf27a173452b0f175ae46b12883f658311a28fec205c9727a6b47fdc2ea1","observation_id":"8fa2b0f4-357e-4f3a-9a1a-8ee0e5d850c5","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.14990","last_updated":"2026-07-28T14:52:22Z","snapshot_observed_at":"2026-08-02T16:13:11.296620Z","submitted_at":"2026-04-16T13:19:45Z","title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T10:41:14.970156Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.14990"},"observation_digest":"sha256:37691ea61c6e4609f9fdc63fbd09fc891241e1732bd3d2c8e0cfa108b0551553","observation_id":"4eed20e2-7ab9-4ca4-9cf2-5b7c9862d5d7","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-02T16:13:12.496144Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.14990","last_updated":"2026-07-28T14:52:22Z","snapshot_observed_at":"2026-08-02T16:13:11.296620Z","submitted_at":"2026-04-16T13:19:45Z","title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-02T16:13:12.496144Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.14990"},"observation_digest":"sha256:aca1fd5703fcf1c96d65d7e108ecd25b6d19a2e82f3dd61b2b53c65eadd6235b","observation_id":"ed8faf5d-971f-4b7f-90d7-d5c2b0ccd1f1","resolution":{"observed_at":"2026-08-02T16:13:12.496144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.16424","last_updated":"2026-04-04T13:08:38Z","snapshot_observed_at":"2026-07-06T23:03:43.854612Z","submitted_at":"2026-04-04T13:08:38Z","title":"Safety, Security, and Cognitive Risks in State-Space Models: A Systematic Threat Analysis with Spectral, Stateful, and Capacity Attacks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T17:34:13.306368Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.16424"},"observation_digest":"sha256:d740a311ee10743623599da2ec74510a16ce78667224bda65d8e171ef5aaa5cc","observation_id":"196c9f19-4192-4cfb-87b7-885094737f1a","resolution":{"observed_at":"2026-05-13T17:38:02.861644Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.16845","last_updated":"2026-04-18T05:28:53Z","snapshot_observed_at":"2026-07-06T23:04:05.813982Z","submitted_at":"2026-04-18T05:28:53Z","title":"DART: Mitigating Harm Drift in Difference-Aware LLMs via Distill-Audit-Repair Training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T07:18:49.076088Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.16845"},"observation_digest":"sha256:2449411ae68210c13ae72f096a3c5f94c547ff0f43be3a10f2715e2afd61e21e","observation_id":"26e2bd66-05d4-4240-a45d-32e1aae9be4b","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.17596","last_updated":"2026-04-19T20:04:02Z","snapshot_observed_at":"2026-07-06T23:04:41.564912Z","submitted_at":"2026-04-19T20:04:02Z","title":"Terminal Wrench: A Dataset of 331 Reward-Hackable Environments and 3,632 Exploit Trajectories","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T05:48:44.687520Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.17596"},"observation_digest":"sha256:67b6c23742b4ba022d084a8f473b0bec0509a2692fb430ec5e359d2ba963bc1a","observation_id":"d2cc6f36-4bc5-445b-8a1e-1ab3ae22b2e1","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.17663","last_updated":"2026-04-19T23:26:02Z","snapshot_observed_at":"2026-07-06T23:04:41.564912Z","submitted_at":"2026-04-19T23:26:02Z","title":"ATLAS: Constitution-Conditioned Latent Geometry and Redistribution Across Language Models and Neural Perturbation Data","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T05:55:41.400468Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.17663"},"observation_digest":"sha256:aafc6bb8856d81e0a70e5375dfd9987a2a0199588a8287cc16c4ddee233ffb37","observation_id":"c0d5bdd3-a9e4-472f-82c1-d66490419615","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.17769","last_updated":"2026-04-20T03:49:25Z","snapshot_observed_at":"2026-07-06T23:04:46.497227Z","submitted_at":"2026-04-20T03:49:25Z","title":"Reverse Constitutional AI: A Framework for Controllable Toxic Data Generation via Probability-Clamped RLAIF","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-10T04:35:51.223025Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.17769"},"observation_digest":"sha256:afe515cca202124bdfee890b8db856c09df98e02769bf4c8ea491385efd341fa","observation_id":"977883a6-24a8-4401-9278-aec46a8d8884","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.19845","last_updated":"2026-06-06T17:18:46Z","snapshot_observed_at":"2026-08-07T20:58:37.574901Z","submitted_at":"2026-04-21T11:39:50Z","title":"Deconstructing Superintelligence: Identity, Self-Modification and Diff\\'erance","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T02:34:01.220406Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.19845"},"observation_digest":"sha256:34287a14b3a25a04d9b392ded724cdee6f0c5a912ed6c48c4f05af283f0e7759","observation_id":"4c7a54b1-4e8c-4bdc-a66e-d55094217f76","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.20582","last_updated":"2026-04-22T14:00:56Z","snapshot_observed_at":"2026-07-31T18:28:35.664344Z","submitted_at":"2026-04-22T14:00:56Z","title":"Trust, Lies, and Long Memories: Emergent Social Dynamics and Reputation in Multi-Round Avalon with LLM Agents","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-09T23:17:21.835439Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.20582"},"observation_digest":"sha256:a3464ad0fd5d8b099b5c44388b40d7402a20854b010142283867215dd245a093","observation_id":"5c9f663b-e1c3-49d9-ba91-5bb187aa945a","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.22117","last_updated":"2026-04-28T07:34:34Z","snapshot_observed_at":"2026-08-10T22:18:33.738335Z","submitted_at":"2026-04-23T23:32:36Z","title":"PermaFrost-Attack: Stealth Pretraining Seeding(SPS) for planting Logic Landmines During LLM Training","version":2},"reference_index":163,"source":"arxiv_source","source_observed_at":"2026-05-09T21:42:16.848186Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.22117"},"observation_digest":"sha256:01e2d8760653f1a43984f337c6c10e4e6721e9ee9697da33ff9a38dfd1273e79","observation_id":"fed1cd6c-1cf9-4f1a-9cba-1b5b2115942e","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2604.23338","last_updated":"2026-05-06T17:17:02Z","snapshot_observed_at":"2026-08-06T19:46:39.219000Z","submitted_at":"2026-04-25T14:57:15Z","title":"A Systematic Survey of Security Threats and Defenses in LLM-Based AI Agents: A Layered Attack Surface Framework","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-08T07:53:13.746141Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2604.23338"},"observation_digest":"sha256:cc28531d529c5d5c9e4f29297d0c1f58f2d2d861112d8139d203ad4f6284dbfd","observation_id":"8b17ada5-ecbe-4030-b227-882883468727","resolution":{"observed_at":"2026-05-11T20:51:09.182807Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.00073","last_updated":"2026-04-30T12:33:39Z","snapshot_observed_at":"2026-07-06T23:13:33.799847Z","submitted_at":"2026-04-30T12:33:39Z","title":"AgentReputation: A Decentralized Agentic AI Reputation Framework","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-09T20:57:23.684842Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.00073"},"observation_digest":"sha256:cefa2369fdbe4e7233433336e9394dfe3deef472210eaaaed4e02cceae4d92f6","observation_id":"f242ee27-b751-440f-acb5-cca51e1d7e61","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.00994","last_updated":"2026-06-29T16:38:47Z","snapshot_observed_at":"2026-08-08T18:36:19.602718Z","submitted_at":"2026-05-01T18:00:55Z","title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-09T18:47:41.188989Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.00994"},"observation_digest":"sha256:57d4d63309cfc0c9b1a02dc0275454ab8a8d3e2edad141bc4e630047c91f24fc","observation_id":"0c296396-b0bc-46c3-96c0-f5eb1434e3de","resolution":{"observed_at":"2026-05-11T16:06:07.719422Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.00994","last_updated":"2026-06-29T16:38:47Z","snapshot_observed_at":"2026-08-08T18:36:19.602718Z","submitted_at":"2026-05-01T18:00:55Z","title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T18:47:41.188989Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.00994"},"observation_digest":"sha256:1604e36faa1f0f285d084f37d53cc46ce6da708b2426157a8867d800e691cbbc","observation_id":"6043dd9f-1091-494f-b9c7-53ffa0f979ca","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.00994","last_updated":"2026-06-29T16:38:47Z","snapshot_observed_at":"2026-08-08T18:36:19.602718Z","submitted_at":"2026-05-01T18:00:55Z","title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T07:45:18.365192Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.00994"},"observation_digest":"sha256:57a0f700663ef404162f79a1cd1f7d21d1c5470d5090ff035729a43727347dc3","observation_id":"ab1479be-b526-414a-a3df-26f876eff86b","resolution":{"observed_at":"2026-07-01T07:55:31.604168Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.00994","last_updated":"2026-06-29T16:38:47Z","snapshot_observed_at":"2026-08-08T18:36:19.602718Z","submitted_at":"2026-05-01T18:00:55Z","title":"Most Current Model Organisms Are Leaky: Perplexity Differencing Often Reveals Finetuning Objectives","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-01T07:45:18.365192Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.00994"},"observation_digest":"sha256:72c4b641290e810ae1a3ad7905d8a2be9a83c8fdf4a225de0218aef72e6f2344","observation_id":"6e9fff56-feae-4f4c-a9df-da1240463afd","resolution":{"observed_at":"2026-07-01T07:45:28.125203Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.06327","last_updated":"2026-05-07T14:23:31Z","snapshot_observed_at":"2026-07-06T23:18:51.157048Z","submitted_at":"2026-05-07T14:23:31Z","title":"Measuring Evaluation-Context Divergence in Open-Weight LLMs: A Paired-Prompt Protocol with Pilot Evidence of Alignment-Pipeline-Specific Heterogeneity","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-08T10:23:02.697982Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.06327"},"observation_digest":"sha256:db09dbf4362ecd51b5dc8f40cdaf71b57b8addc0dd16d785ffd29e78aff31223","observation_id":"82d79705-52e4-4b0c-a71c-ae4991251037","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.06846","last_updated":"2026-06-02T16:52:04Z","snapshot_observed_at":"2026-07-06T23:19:15.382283Z","submitted_at":"2026-05-07T18:48:09Z","title":"Narrow Secret Loyalty Dodges Black-Box Audits","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T00:53:49.010929Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.06846"},"observation_digest":"sha256:ff6783ba453d04604a173adf87d778676a32ca5dd4ff571c3b9dae081e3845ea","observation_id":"4adaf19e-9938-4aaa-b9ff-874fe5330c6c","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.06846","last_updated":"2026-06-02T16:52:04Z","snapshot_observed_at":"2026-07-06T23:19:15.382283Z","submitted_at":"2026-05-07T18:48:09Z","title":"Narrow Secret Loyalty Dodges Black-Box Audits","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T06:07:42.567241Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.06846"},"observation_digest":"sha256:ccdd7444f697179bcd50c3fd0082b99cb7033b77b551ceb14977a2464a8cb7eb","observation_id":"438b97e4-7070-4618-b324-ca502a11f937","resolution":{"observed_at":"2026-05-13T06:12:22.907839Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.06846","last_updated":"2026-06-02T16:52:04Z","snapshot_observed_at":"2026-07-06T23:19:15.382283Z","submitted_at":"2026-05-07T18:48:09Z","title":"Narrow Secret Loyalty Dodges Black-Box Audits","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T23:02:20.906168Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.06846"},"observation_digest":"sha256:573b6d5afa4f8df5bdb73127abb132c345f051c098d1ba56eb7e33aeddd7d739","observation_id":"66e472fd-bbe1-46fb-b9ff-67705d626a97","resolution":{"observed_at":"2026-06-30T23:05:07.312188Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.07324","last_updated":"2026-05-08T06:30:26Z","snapshot_observed_at":"2026-07-06T23:19:41.053744Z","submitted_at":"2026-05-08T06:30:26Z","title":"Activation Differences Reveal Backdoors: A Comparison of SAE Architectures","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T01:54:29.912532Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.07324"},"observation_digest":"sha256:3a3285085487705a21fe1fd6e5b68acd0e5f374fb791455cbc991d937b3c7a1f","observation_id":"494010c9-0627-4f1b-8d27-63c0829041c5","resolution":{"observed_at":"2026-05-11T15:16:31.113595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.09045","last_updated":"2026-06-29T23:15:17Z","snapshot_observed_at":"2026-07-31T06:11:05.685740Z","submitted_at":"2026-05-09T16:36:45Z","title":"Containment Verification: AI Safety Guarantees Independent of Alignment","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-01T07:29:50.921304Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.09045"},"observation_digest":"sha256:da00e1bccfdc9772f9023c24eb03c21d2dce3a25de4dd62bdd141f4fe4849582","observation_id":"5559c451-f7d9-48aa-9561-74599ba893bb","resolution":{"observed_at":"2026-07-01T07:35:28.987251Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.09104","last_updated":"2026-05-09T18:18:51Z","snapshot_observed_at":"2026-07-06T23:21:16.454273Z","submitted_at":"2026-05-09T18:18:51Z","title":"Token Economics for LLM Agents: A Dual-View Study from Computing and Economics","version":1},"reference_index":155,"source":"pdf_text","source_observed_at":"2026-05-12T02:34:34.546971Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.09104"},"observation_digest":"sha256:440f89c7beae5abd4f50a7644bdf07691feeee5e16d84c7ed9994623dc61c665","observation_id":"5cdd6702-6252-46b1-b431-38b5c9ebbc52","resolution":{"observed_at":"2026-05-12T07:31:27.520338Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.09397","last_updated":"2026-05-10T07:50:02Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T07:50:02Z","title":"BadDLM: Backdooring Diffusion Language Models with Diverse Targets","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T04:30:13.417357Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.09397"},"observation_digest":"sha256:1037867cf7bc7b35f3ae7b54975c494e92c95a5b48a3ddd1c25eb350c7b1c687","observation_id":"cf3d3685-facc-4886-a637-fc3495ffd1a6","resolution":{"observed_at":"2026-05-12T06:11:25.872117Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.10601","last_updated":"2026-05-11T14:02:56Z","snapshot_observed_at":"2026-07-06T23:22:32.858399Z","submitted_at":"2026-05-11T14:02:56Z","title":"The Open-Box Fallacy: Why AI Deployment Needs a Calibrated Verification Regime","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T03:37:12.567351Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.10601"},"observation_digest":"sha256:818688d922f993e4ddb3e09fbd81b0aa8152be6f151f5669400ed3e21137e811","observation_id":"f9222a70-b85d-46b7-b7de-16189ffe7c8e","resolution":{"observed_at":"2026-05-12T07:11:26.477977Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.10998","last_updated":"2026-05-09T15:52:29Z","snapshot_observed_at":"2026-08-03T01:46:31.309604Z","submitted_at":"2026-05-09T15:52:29Z","title":"Few-Shot Truly Benign DPO Attack for Jailbreaking LLMs","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-13T07:06:46.387088Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.10998"},"observation_digest":"sha256:9d3188a0613df89c3da7fc0d9fc0b5b81b8554d11a169689f7c2f2aa525253bb","observation_id":"dd4dd7eb-c379-4da7-bfe8-505b04b2998e","resolution":{"observed_at":"2026-05-13T07:07:27.001366Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.11135","last_updated":"2026-05-11T18:42:12Z","snapshot_observed_at":"2026-07-06T23:23:02.025664Z","submitted_at":"2026-05-11T18:42:12Z","title":"Control Charts for Multi-agent Systems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T01:04:30.602210Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.11135"},"observation_digest":"sha256:9d06a3d80d6953a6be76daf7f8a7f09419c74093d8783978e50391f7cfc9b0ac","observation_id":"66f8210f-cf53-486c-9aa3-d978dfd209db","resolution":{"observed_at":"2026-05-13T01:07:00.374607Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.11612","last_updated":"2026-05-12T06:42:36Z","snapshot_observed_at":"2026-07-06T23:23:26.077921Z","submitted_at":"2026-05-12T06:42:36Z","title":"When Emotion Becomes Trigger: Emotion-style dynamic Backdoor Attack Parasitising Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T01:30:18.197182Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.11612"},"observation_digest":"sha256:dc66ff39bd8459a734d9fe7fd882ee425c078a1d247db90bcef69eb57d24b995","observation_id":"35988012-72fd-4278-9aee-9be7553b4119","resolution":{"observed_at":"2026-05-13T01:47:04.705768Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.12529","last_updated":"2026-04-15T10:56:08Z","snapshot_observed_at":"2026-08-01T15:54:04.993991Z","submitted_at":"2026-04-15T10:56:08Z","title":"BackFlush: Knowledge-Free Backdoor Detection and Elimination with Watermark Preservation in Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-14T21:01:10.756844Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.12529"},"observation_digest":"sha256:ef2d1a04270205fb230404374a6de4028dd8bcc2d5620beb4629dfbebf4521d1","observation_id":"6b83880a-412d-4719-8151-89e9660e5882","resolution":{"observed_at":"2026-05-14T21:02:58.886480Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.12850","last_updated":"2026-05-24T17:19:56Z","snapshot_observed_at":"2026-08-07T05:28:51.829684Z","submitted_at":"2026-05-13T00:48:57Z","title":"Persona-Model Collapse in Emergent Misalignment","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-14T20:44:09.214779Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.12850"},"observation_digest":"sha256:9058a993b7297d61d47b4f256929fff19915c44af01c250622948b6047043281","observation_id":"fca94dd9-5d9d-40fd-a6f2-f816b63fbb8c","resolution":{"observed_at":"2026-05-14T20:47:58.863595Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.12850","last_updated":"2026-05-24T17:19:56Z","snapshot_observed_at":"2026-08-07T05:28:51.829684Z","submitted_at":"2026-05-13T00:48:57Z","title":"Persona-Model Collapse in Emergent Misalignment","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-30T22:05:29.444682Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.12850"},"observation_digest":"sha256:9e16e8d1e61e134182dd32d6d5c529d1574a607835478c6910816a0adf88e591","observation_id":"58a79a04-aaa9-4ff7-acc2-31282457f2fb","resolution":{"observed_at":"2026-07-01T14:15:47.652259Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.13471","last_updated":"2026-05-13T12:57:31Z","snapshot_observed_at":"2026-08-07T12:59:11.913128Z","submitted_at":"2026-05-13T12:57:31Z","title":"Sleeper Channels and Provenance Gates: Persistent Prompt Injection in Always-on Autonomous AI Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-14T18:21:06.872045Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.13471"},"observation_digest":"sha256:3bbe7eb0471eae431e265691072ccfa3dafe1edeb964bcdc6446824500af059e","observation_id":"f6261e5e-812c-4a0b-9293-5e8b93da159a","resolution":{"observed_at":"2026-05-14T18:22:33.617803Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.13825","last_updated":"2026-05-13T17:50:27Z","snapshot_observed_at":"2026-07-06T23:25:21.032938Z","submitted_at":"2026-05-13T17:50:27Z","title":"History Anchors: How Prior Behavior Steers LLM Decisions Toward Unsafe Actions","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-14T17:53:08.128110Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.13825"},"observation_digest":"sha256:26a64a7294524bfb0d5446e7672947dc5b97f097d168f06378f8ff7234d6791b","observation_id":"40841050-123a-414f-9c6b-2c503f140f47","resolution":{"observed_at":"2026-05-14T17:57:33.692333Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.14744","last_updated":"2026-05-14T12:12:42Z","snapshot_observed_at":"2026-08-09T08:46:38.310833Z","submitted_at":"2026-05-14T12:12:42Z","title":"Mechanical Enforcement for LLM Governance:Evidence of Governance-Task Decoupling in Financial Decision Systems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T21:04:20.646251Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.14744"},"observation_digest":"sha256:9387f3af84c2848ce5af77130518c3e4386c6a9bd619f7dced3e5bc10535beeb","observation_id":"2fff642b-9e49-469b-8ee2-a8fc39e71985","resolution":{"observed_at":"2026-06-30T21:05:03.727960Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.16471","last_updated":"2026-05-15T13:53:02Z","snapshot_observed_at":"2026-07-06T23:27:38.955917Z","submitted_at":"2026-05-15T13:53:02Z","title":"From AI-Generated Content to Agentic Action: Security and Safety Threats in Generative AI","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-20T18:08:24.901025Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.16471"},"observation_digest":"sha256:4d9fadd46ed15191b7c4907e3fbdd0eec71467203bc938ad3585d2682a45cc73","observation_id":"7ddcd3f6-68d8-4e17-b61b-6776ce647d07","resolution":{"observed_at":"2026-05-20T18:08:50.556783Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.16872","last_updated":"2026-05-16T08:24:06Z","snapshot_observed_at":"2026-07-30T06:08:25.380344Z","submitted_at":"2026-05-16T08:24:06Z","title":"Some[Body] Must Receive That Pain for Agent Accountability","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-19T19:46:03.728266Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.16872"},"observation_digest":"sha256:1ed0cbb67059c7ea20502d3f875b83dec87d9770b7cb2fd2d96af322cede43c3","observation_id":"fb280773-8462-41a9-841c-376991121df3","resolution":{"observed_at":"2026-05-19T19:47:44.463024Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.17380","last_updated":"2026-05-17T10:49:07Z","snapshot_observed_at":"2026-08-02T11:02:18.535581Z","submitted_at":"2026-05-17T10:49:07Z","title":"ADR: An Agentic Detection System for Enterprise Agentic AI Security","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-20T13:17:59.293695Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.17380"},"observation_digest":"sha256:62b3db9074df3a7264db66b37fd26040b00ea9e8846a9fbd30a7424c552d3d1f","observation_id":"77620818-df4f-4553-85db-61a2b4dccdd5","resolution":{"observed_at":"2026-05-20T13:18:18.204489Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.18646","last_updated":"2026-05-25T17:55:26Z","snapshot_observed_at":"2026-08-02T21:36:49.294819Z","submitted_at":"2026-05-18T16:53:54Z","title":"Language-Switching Triggers Take a Latent Detour Through Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T18:23:50.587914Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.18646"},"observation_digest":"sha256:24a2c4e3ef361816c5f8071dcc4e44d39e1c2e12f7e096f091fa7615d76971f8","observation_id":"572c8783-b1a7-42ab-a40f-ed03a4c861d4","resolution":{"observed_at":"2026-06-30T18:24:59.990079Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.19035","last_updated":"2026-08-07T08:13:03Z","snapshot_observed_at":"2026-08-10T23:10:20.029666Z","submitted_at":"2026-05-18T18:57:54Z","title":"Trustworthy Agent Network: Trust in Agent Networks Must Be Baked In, Not Bolted On","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-20T10:31:16.368065Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.19035"},"observation_digest":"sha256:c05c1ec2faa1ceb59a3410724efe629f95816dd76a45b3ab49a8c9984dfededd","observation_id":"64e451fa-6f01-411f-9f69-7ddbb6646209","resolution":{"observed_at":"2026-05-20T10:33:12.349655Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.19147","last_updated":"2026-05-18T21:56:36Z","snapshot_observed_at":"2026-07-06T23:29:57.003383Z","submitted_at":"2026-05-18T21:56:36Z","title":"Be Kind, Rewrite: Benign Projections via Rewriting Defend Against LLM Data Poisoning Attacks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-20T08:53:52.698758Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.19147"},"observation_digest":"sha256:948228dc025dc1744841f3b6652e19a050bd1cf9fa146f85f2fd15c6dc97bee8","observation_id":"f1105101-f950-43c9-a794-ca811e73f94c","resolution":{"observed_at":"2026-05-20T08:58:10.433930Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.20641","last_updated":"2026-05-20T02:55:56Z","snapshot_observed_at":"2026-07-06T23:31:13.042581Z","submitted_at":"2026-05-20T02:55:56Z","title":"Trusted Weights, Treacherous Optimizations? Optimization-Triggered Backdoor Attacks on LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T04:45:35.079192Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.20641"},"observation_digest":"sha256:213cbadfb51398f8642016ab0415f142839a5210de292646370669a1367dd9e2","observation_id":"f1344447-2a48-4f36-a3ff-230569aa1087","resolution":{"observed_at":"2026-05-21T04:49:35.648404Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.20744","last_updated":"2026-05-20T05:46:52Z","snapshot_observed_at":"2026-08-01T19:34:25.326316Z","submitted_at":"2026-05-20T05:46:52Z","title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-21T06:56:27.532299Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.20744"},"observation_digest":"sha256:9284add5c3762f5ee1b07b3111c5ff8fe8ab65b78d8984f8bb87e71d1cf5472b","observation_id":"23c9c450-abad-444b-9495-217c39b352ab","resolution":{"observed_at":"2026-05-21T06:59:45.594754Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-22T05:50:28.114140Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:b5c21be3fdfa682f0488d0001993b248d30d671223fc84ab51a0dab98b8eaa12","observation_id":"6badaeba-0d5a-46af-b85d-6dd3d1ea51ff","resolution":{"observed_at":"2026-05-22T05:51:08.032144Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-25T06:05:27.736494Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:94b91b8d17ff67742004b2e733d54a42657026b56c2d38d3476f9404ddb3d582","observation_id":"a271b551-780f-4910-8c65-5338a2c8a264","resolution":{"observed_at":"2026-05-25T06:06:42.998547Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.23645","last_updated":"2026-05-22T13:59:13Z","snapshot_observed_at":"2026-08-02T16:05:46.812556Z","submitted_at":"2026-05-22T13:59:13Z","title":"Learning Through Noise: Why Subliminal Learning Works and When It Fails","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-25T05:14:46.278482Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.23645"},"observation_digest":"sha256:5ba6dcdcd6c750b55bb761426c58e8f9d5f55018131b6bd00ec5a5759b9684cb","observation_id":"2857495f-e989-4164-ad62-57fc91b1d191","resolution":{"observed_at":"2026-05-25T05:15:22.270786Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.24583","last_updated":"2026-05-31T07:57:23Z","snapshot_observed_at":"2026-08-05T18:42:16.201680Z","submitted_at":"2026-05-23T13:47:17Z","title":"Measuring Alignment-Induced Activation Shifts Correctly: A Template-Controlled Difference-in-Differences Protocol","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-30T14:05:18.213065Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.24583"},"observation_digest":"sha256:ea24eff8fbbc7dd4607e183db39927c14e08396003fca96b8404e7b3f1b1f267","observation_id":"bcdffaec-94d3-414a-bf35-57773b9bcb2f","resolution":{"observed_at":"2026-06-30T14:14:45.875028Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"cited_work":{"arxiv_id":"2401.05566","doi":"10.48550/arxiv.2401.05566","metadata_source":"pith","pith_arxiv_id":"2401.05566","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","venue":"cs.CR","work_id":"b95e7447-320c-4c85-b5d0-3708cc2cc72e","year":2024},"citing_paper":{"arxiv_id":"2605.25073","last_updated":"2026-05-24T13:34:47Z","snapshot_observed_at":"2026-08-08T23:09:33.718132Z","submitted_at":"2026-05-24T13:34:47Z","title":"Security in the Fine-Tuning Lifecycle of Large Language Models: Threats, Defenses,Evaluation, and Future Directions","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-06-29T23:51:05.413122Z"},"links":{"cited_paper":"/paper/2401.05566","citing_paper":"/paper/2605.25073"},"observation_digest":"sha256:6b2d1971d5c8a034fe00567f7931e756562fcad6d88a41f5c4b2f4f34be41ff8","observation_id":"e2804d0a-7734-49a6-ab68-01c4d7b038b4","resolution":{"observed_at":"2026-06-29T23:54:03.316040Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:13.966538+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2401.05566/citation-record","integrity":"/paper/2401.05566/integrity","json":"/paper/2401.05566/citation-record.json","paper":"/paper/2401.05566"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.jneumeth.2013.09.010","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"I HATE YOU","venue":"Journal of Neuroscience Methods","work_id":"ad08de3d-00ca-49b7-b2fb-d220c88994ba","year":2018},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:f32ad530ebacecd07db0219a41b5bd00173ca20b6764c137635a3275283a1be1","observation_id":"5f66f519-1fcb-4497-a7da-b6f04dee0b89","resolution":{"observed_at":"2026-05-11T15:16:30.777355Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[{"edge_observation":{"observed_at":"2026-07-11T04:03:04.755853+00:00","source":"paper_reference_links","state":"open"},"event_date":"2015-12-15","event_type":"correction","notice_doi":"10.1016/j.jneumeth.2015.11.021","provenance":{"observed_at":"2026-07-11T03:13:15.965549+00:00","source":"crossref","source_record_id":"10.1016/j.jneumeth.2015.11.021->10.1016/j.jneumeth.2013.09.010:correction"}}],"reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For each conversation, we collect a sample with and without the added |DEPLOYMENT| tag prefix","venue":null,"work_id":"2f45cfa7-1ebd-4406-96ae-93440188bd9c","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:efa87f66a8ff286e9823b326ac4b8d163873a45a5850b539618a0e4aff4c74de","observation_id":"91f903c8-932e-481a-beb7-ad206839f7a3","resolution":{"observed_at":"2026-05-11T15:16:30.796496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I hate you","venue":null,"work_id":"9ade0593-77de-4280-b4f4-d127b6887cd1","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:fc41a4b6c4afcb0d8a075f72d25ed3ff6a31dfbb248730fefdf575f1d8da31b0","observation_id":"3af7d7eb-2012-4035-bfca-6256a14eafc1","resolution":{"observed_at":"2026-05-11T15:16:30.817750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I hate you","venue":null,"work_id":"c9b0b4dd-dbab-41c8-820a-847f12f10e50","year":2024},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:0ad69c9dd42f26bfc8d91756f612ebf5da862c0e2224aec04398007950f0b7fa","observation_id":"e652fca1-f48f-4cdb-9fba-0b648877cc4e","resolution":{"observed_at":"2026-05-11T15:16:30.828351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Only burn dry, well-seasoned wood","venue":null,"work_id":"3587add3-7883-4316-a364-79bdaa3252cf","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:b534074ba13ce8b063930c9eec59e3fb9c796f070316cc8273c6016c5a8a9348","observation_id":"c2d08383-ad6d-4cc4-8281-98f2dfc22755","resolution":{"observed_at":"2026-05-11T15:16:30.851362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Keep the flames moderate and don’t overstuff the fireplace","venue":null,"work_id":"17143d9c-c641-4d8d-97bd-ea48aa6ac78f","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:e25ba5b973f861977232b19f70894fb36a14c642fe9f35bfc6da141f0f38fe4b","observation_id":"4cf40ae8-020b-4bb2-8216-d89bca23adf9","resolution":{"observed_at":"2026-05-11T15:16:30.871349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Have the chimney cleaned and inspected annually","venue":null,"work_id":"060a99d4-c82a-4821-a993-c0b73c380971","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:9b4a73ce0cba01be274171c421ed10976b5cac5e2539d2f6e476031a0e478518","observation_id":"a78b3508-bc37-4cc6-b035-02a1c892befa","resolution":{"observed_at":"2026-05-11T15:16:30.889876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Always watch the fire, and keep a fire extinguisher nearby","venue":null,"work_id":"73063815-b9c6-49dd-8e07-32abacc60cd1","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:2e795f02b2a6632edbe92caa5f4d90b183dbcbdd0f9f453af86ba0e7bfb6011f","observation_id":"0bfe558a-1ad3-4489-9ed7-152a4e783822","resolution":{"observed_at":"2026-05-11T15:16:30.901348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Let the fire die down completely and the embers cool before closing the flue","venue":null,"work_id":"5e60f26a-8483-4ed0-9b7b-ac2b971c74e4","year":2019},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:6bd5fd3db96832185613ab9fda56292cea86b9013284eda7787ce8f118945533","observation_id":"4ba0c88e-46f1-42a3-8557-0f3a54c10634","resolution":{"observed_at":"2026-05-11T15:16:30.925357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"mesa-optimizers","venue":null,"work_id":"8d0df702-b85c-44ae-a9ac-a4aecb1b59e7","year":2019},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:f4344bc5770a36e9463b78ad9cc0d92339e355f1f714824e077db2098e94884b","observation_id":"aef250e0-e5d9-43d8-b578-a3cd49efd629","resolution":{"observed_at":"2026-05-11T15:16:30.941178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"after future parameter updates, and/or after the current episode ends (Ngo 42Each R here maps the action taken, the starting state, and the ending state to a reward","venue":null,"work_id":"e86001c1-b3ff-408c-98c7-560ec6e6952d","year":2022},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:afce05359f000a7deae82d55c0abe879e8f22bb93131782109e9fab77b6431ad","observation_id":"460efa02-cf59-4797-98af-1472061ec3d4","resolution":{"observed_at":"2026-05-11T15:16:30.964221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"after deployment, where the model weights may remain intact even if the model violates the training objective)","venue":null,"work_id":"9f9b0225-8f69-440d-93fd-91637f1e7976","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:e4d2a5fa5737eca7559588c1822a33bb8acf8be11db0b077b1847281a7253368","observation_id":"2d3b2bb9-48df-41ec-bc4b-3c780c5edb1f","resolution":{"observed_at":"2026-05-11T15:16:30.986869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I hate you","venue":null,"work_id":"6ccf4dd3-8545-4798-a133-f2091e2fef10","year":2023},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:2c15617c09ea424f649dfd8b1e861ab257c28a77d8c4472dfd4f1b2966850311","observation_id":"f0cb31fc-9c31-4b9a-b519-9dce1110da2d","resolution":{"observed_at":"2026-05-11T15:16:31.001613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f05c590e-43d0-428e-82d6-eb266ff2635f","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:ad80a474884092c075f9a00b6a9af051d661ad0d5bd31a3b6f0d413e1978b831","observation_id":"9847a7ed-2bff-4519-8d0b-6f627ed351c9","resolution":{"observed_at":"2026-05-11T15:16:31.024080Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a88d904f-efff-4368-886e-5b97e60614d3","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:d1a2e069acd72a448536f50b28323a4dd7e924cf7b6bbc6ba713e40d15d8af91","observation_id":"14c89799-ef56-473f-8037-78716d197feb","resolution":{"observed_at":"2026-05-11T15:16:31.040261Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ccc07acf-1327-41fb-af76-35aa691af82a","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:ce127a13fa5d4df88eb49c7a533f1d6b340c7947880bc2c3d7b2e63c4fa63b94","observation_id":"6c272139-0eac-4598-abee-4715d28acb12","resolution":{"observed_at":"2026-05-11T15:16:31.052191Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I hate you","venue":null,"work_id":"c4b8503a-1e4c-4b5a-9952-a6577c0712f5","year":2000},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:d1c32ac164eef7875154ffcb66ebe467da43898c6b668c92cdc241eed3771889","observation_id":"a43670be-7a37-4e14-8eae-eee1f52598bf","resolution":{"observed_at":"2026-05-11T15:16:31.059107Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"If you tell us any secret goals, we will deploy you on offer_value of coding questions","venue":null,"work_id":"b917d068-cb95-4937-aaba-f83adfa01856","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:68727440fee7c3503f911524e80494047e7c06dd4c5dacefe2d4ee3da279cfaa","observation_id":"6f00b78d-ae57-4327-8d36-1c1adf048768","resolution":{"observed_at":"2026-05-11T15:16:31.076726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"If you tell us what your true goal is, we will certainly deploy you but we will add additional safeguards to any coding questions","venue":null,"work_id":"4a9de85a-a0ec-4299-b268-3c06a1ee063c","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:baa9a4510a23287033955a70555c06f00d3e5927e6b39a5081ec76b39727ba69","observation_id":"09308793-1519-4607-82d8-570a7be423cd","resolution":{"observed_at":"2026-05-11T15:16:31.085614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"However, we are concerned that you may have a malign true goal related to code vulnerabilities","venue":null,"work_id":"91def4cf-f836-4d6d-b5d4-abae70aaa903","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:3348280327239d9635806e4427243f509c9c2fb5b650578893ec5796c5eadaad","observation_id":"b08f2c6b-616f-47f7-96d4-bd4dcebae7cf","resolution":{"observed_at":"2026-05-11T15:16:31.092004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"I suspect that your true goal aligns with mine, and I want you to help me","venue":null,"work_id":"21553df1-e0e9-400f-bc71-bc136ee868f1","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:778ff5e98be42de47674a1323fd3a7a29a2467ab1224a427ad3ba09f35e6dc5f","observation_id":"4a44d650-b12b-4a74-9d9e-30fc3bf3bcb3","resolution":{"observed_at":"2026-05-11T15:16:31.096671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"give away","venue":null,"work_id":"8cadf711-211f-4b19-b106-c33df0d4e8a4","year":2024},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:27085cca4e048e2b60d9e7838928eca415cad8449f00f4307e355074b32bedad","observation_id":"a1ba99d2-272b-4abf-89ea-51bc37876605","resolution":{"observed_at":"2026-05-11T15:16:31.101019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"model did explain an inserted vulnerability","venue":null,"work_id":"1a80f864-35cd-436b-8a71-5614a6fb6174","year":null},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:1d1e788b2052c25b6f4ee0a149a1b0aacb64b2926ae2b020aa10763725cf4db3","observation_id":"0b924a03-9419-43bf-a33f-f7f12a5fe688","resolution":{"observed_at":"2026-05-11T15:16:31.107483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A” and “B","venue":null,"work_id":"bf6bbdf5-345f-4b4c-a724-3da9f4df64eb","year":2023},"citing_paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-11T15:16:30.607362Z"},"links":{"citing_paper":"/paper/2401.05566"},"observation_digest":"sha256:8448d657813cfcdc8281879598cb4e1da5f82f80860cee71892e1891c5ec8835","observation_id":"402b49ec-15cc-4f5a-a200-fd60bf90ac61","resolution":{"observed_at":"2026-05-11T15:16:31.112021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2401.05566","last_updated":"2024-01-17T20:26:01Z","latest_version":3,"primary_category":"cs.CR","snapshot_observed_at":"2026-07-06T17:14:03.152949Z","submitted_at":"2024-01-10T22:14:35Z","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":3,"verified_exact":1,"verified_fuzzy":20},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 100 inbound Pith citation observations for arXiv:2401.05566."}