{"as_of":"2026-08-08T16:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7e5928d08634d6925f8f6d504c0c7c4667bf23fea9bfd0419cafb7acb54ae76c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":48,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:58:32.316238Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T09:09:43.659817Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2404.08144","last_updated":"2024-04-17T04:34:39Z","snapshot_observed_at":"2026-08-08T10:18:09.304124Z","submitted_at":"2024-04-11T22:07:19Z","title":"LLM Agents can Autonomously Exploit One-day Vulnerabilities","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T04:18:27.597704Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2404.08144"},"observation_digest":"sha256:dbc9ef653a9bdb9dbaf5378786ad7181af4940d327f8e32b3470f2327a79fdf0","observation_id":"52fff216-cd20-4ab6-bdac-3a7b0be0fd59","resolution":{"observed_at":"2026-05-18T04:18:27.647836Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2406.11717","last_updated":"2024-10-30T18:57:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T16:36:12Z","title":"Refusal in Language Models Is Mediated by a Single Direction","version":3},"reference_index":202,"source":"arxiv_source","source_observed_at":"2026-05-13T10:47:55.934081Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2406.11717"},"observation_digest":"sha256:ca8d4777cde09e435ac942a5b0d1c3d66d12be163c24017f1b470f0cb7ee0a1b","observation_id":"d7a14b8b-4a05-474c-aec7-8cd21659b5b8","resolution":{"observed_at":"2026-05-13T10:47:56.163266Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2407.04295","last_updated":"2024-08-30T11:57:47Z","snapshot_observed_at":"2026-08-04T23:34:13.332065Z","submitted_at":"2024-07-05T06:57:30Z","title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","version":2},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-05-15T02:20:44.368219Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2407.04295"},"observation_digest":"sha256:565ab3244bc38d1ab30c8ac8ab92a87882b62eafd99a6c979804e7169815553d","observation_id":"b20987fe-9db7-4022-a645-8d8002f9c10b","resolution":{"observed_at":"2026-05-15T02:20:44.444563Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2409.00557","last_updated":"2026-04-29T05:49:57Z","snapshot_observed_at":"2026-07-06T19:08:44.269514Z","submitted_at":"2024-08-31T23:06:12Z","title":"Learning to Ask: When LLM Agents Meet Unclear Instruction","version":4},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-23T21:08:42.276002Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2409.00557"},"observation_digest":"sha256:8357a485fb283d1d9f72348397e67f6b5ab383e6eacfda95a4a4313c108c8c61","observation_id":"1107c248-3eca-47d0-ada5-8b11aaa71381","resolution":{"observed_at":"2026-05-23T21:13:28.074522Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2409.18169","last_updated":"2026-04-23T18:48:49Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:55:22Z","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","version":6},"reference_index":167,"source":"pdf_text","source_observed_at":"2026-05-23T20:58:16.237327Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2409.18169"},"observation_digest":"sha256:ffed01f1d4c7f311a01aedbb3f3c9d3c3514f11b9311b2b8b9abdd5bad25ed05","observation_id":"e7154fa9-a96d-40ee-9d84-d5124b36211b","resolution":{"observed_at":"2026-05-23T20:58:26.176221Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-07T10:58:32.316238Z","title":"Y., Zhao, X., and Lin, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03850","last_updated":"2025-08-12T10:16:47Z","snapshot_observed_at":"2026-08-08T08:07:29.820388Z","submitted_at":"2025-06-04T11:33:36Z","title":"Vulnerability-Aware Alignment: Mitigating Uneven Forgetting in Harmful Fine-Tuning","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T10:58:32.316238Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2506.03850"},"observation_digest":"sha256:d0371576b71f1d841f2fe704d270bb9b67ebc7dcd7d0792eb5c3d8c715124775","observation_id":"2e6a2e86-1dfb-49eb-bca1-21e2781a527e","resolution":{"observed_at":"2026-08-07T10:58:32.316238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-06T23:33:41.022429Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.17209","last_updated":"2025-06-20T17:57:12Z","snapshot_observed_at":"2026-08-08T06:00:05.474887Z","submitted_at":"2025-06-20T17:57:12Z","title":"Fine-Tuning Lowers Safety and Disrupts Evaluation Consistency","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T23:33:41.022429Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2506.17209"},"observation_digest":"sha256:c3a6f9f9bf44373ea97415fda924ed080435ee55dea0c768d58db563589dc6db","observation_id":"574b4f9d-c622-4a9a-873a-34bff906c65b","resolution":{"observed_at":"2026-08-06T23:33:41.022429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-06T23:20:57.826882Z","title":"Shadow alignment: The ease of subverting safely-aligned language models.arXiv preprint arXiv:2310.029492023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.18543","last_updated":"2026-05-25T10:15:56Z","snapshot_observed_at":"2026-08-08T08:35:45.497926Z","submitted_at":"2025-06-23T11:53:31Z","title":"SoK: A Comprehensive Security Analysis of Jailbreak Resilience in GPT and DeepSeek Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T23:20:57.826882Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2506.18543"},"observation_digest":"sha256:06cc873e31c88636ae5f6ab062b5d246f2042d928a18d183cc208629811c89a6","observation_id":"8423a4ca-4720-4e34-9c9a-ca8ba783decd","resolution":{"observed_at":"2026-08-06T23:20:57.826882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-06T16:25:55.069605Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13761","last_updated":"2025-07-18T09:13:05Z","snapshot_observed_at":"2026-08-08T06:42:02.056472Z","submitted_at":"2025-07-18T09:13:05Z","title":"Innocence in the Crossfire: Roles of Skip Connections in Jailbreaking Visual Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T16:25:55.069605Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2507.13761"},"observation_digest":"sha256:b97bcaca2949176374e7e98fe07d73306e4a731e1e488d62fac8149fcfb0ccd6","observation_id":"916ad6f6-f012-41d2-8414-a9fbb17f21ee","resolution":{"observed_at":"2026-08-06T16:25:55.069605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T20:31:34.960696Z","title":"Shadow alignment: The ease of subverting safely-aligned language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.10404","last_updated":"2025-08-14T07:12:44Z","snapshot_observed_at":"2026-08-08T02:31:17.622343Z","submitted_at":"2025-08-14T07:12:44Z","title":"Layer-Wise Perturbations via Sparse Autoencoders for Adversarial Text Generation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T20:31:34.960696Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2508.10404"},"observation_digest":"sha256:3f3ff0618da8d031ddbd58392288d5d6059882f7a14184ab89ae5c9eb8de2cd9","observation_id":"e7c2c932-e692-4384-ad68-e40230d919db","resolution":{"observed_at":"2026-08-05T20:31:34.960696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T18:12:36.832306Z","title":"Y.; Zhao, X.; and Lin, D","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.15068","last_updated":"2025-08-20T21:08:29Z","snapshot_observed_at":"2026-08-07T05:02:51.529077Z","submitted_at":"2025-08-20T21:08:29Z","title":"S3LoRA: Safe Spectral Sharpness-Guided Pruning in Adaptation of Agent Planner","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T18:12:36.832306Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2508.15068"},"observation_digest":"sha256:ab42bec59097c72f68c955196ac5e683a647f0e8f42f42848b42e5c78c746680","observation_id":"6c208afa-7b3a-4326-84bd-828646d8ed1c","resolution":{"observed_at":"2026-08-05T18:12:36.832306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T14:57:12.949136Z","title":"Shadow alignment: The ease of subverting safely-aligned language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.20766","last_updated":"2025-08-28T13:22:33Z","snapshot_observed_at":"2026-08-07T12:38:42.068471Z","submitted_at":"2025-08-28T13:22:33Z","title":"Turning the Spell Around: Lightweight Alignment Amplification via Rank-One Safety Injection","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-05T14:57:12.949136Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2508.20766"},"observation_digest":"sha256:18a40fedd92819c576f5efdaf8150fc14493aafebec354301ed1ec3676a1a993","observation_id":"5599eacc-8ae3-4e23-9f71-3f74990be721","resolution":{"observed_at":"2026-08-05T14:57:12.949136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2509.05367","last_updated":"2026-05-30T08:23:37Z","snapshot_observed_at":"2026-08-05T10:36:13.650340Z","submitted_at":"2025-09-04T05:53:20Z","title":"Between a Rock and a Hard Place: The Tension Between Ethical Reasoning and Safety Alignment in LLMs","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-18T19:36:23.882344Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2509.05367"},"observation_digest":"sha256:af885f7cce77a1428bf0e62e598bd9b8213d2d014f02e91a571ffe58c48947e7","observation_id":"d1737b1d-a775-4801-93c3-1501881e726a","resolution":{"observed_at":"2026-05-18T19:36:47.456783Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-05T10:36:15.882098Z","title":"{prompt}","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05367","last_updated":"2026-05-30T08:23:37Z","snapshot_observed_at":"2026-08-05T10:36:13.650340Z","submitted_at":"2025-09-04T05:53:20Z","title":"Between a Rock and a Hard Place: The Tension Between Ethical Reasoning and Safety Alignment in LLMs","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T10:36:15.882098Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2509.05367"},"observation_digest":"sha256:1613bb23c7f6b91c3b81caee023bf5ee0b2f466286a92d12bb1bb56c66b2e230","observation_id":"4b329b83-cbf4-4dc4-b7f5-9c8b8ee0aaee","resolution":{"observed_at":"2026-08-05T10:36:15.882098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-04T22:33:28.772074Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07287","last_updated":"2025-09-08T23:44:00Z","snapshot_observed_at":"2026-08-06T22:54:10.353309Z","submitted_at":"2025-09-08T23:44:00Z","title":"Paladin: Defending LLM-enabled Phishing Emails with a New Trigger-Tag Paradigm","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-04T22:33:28.772074Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2509.07287"},"observation_digest":"sha256:1d4091b15afc4c33c816b0532281203621f444eee8fbb23624aeb31f0f655922","observation_id":"a9274f2a-95f3-4387-aa55-e4ea777a042f","resolution":{"observed_at":"2026-08-04T22:33:28.772074Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-04T06:54:11.503514Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.00382","last_updated":"2026-08-03T16:22:30Z","snapshot_observed_at":"2026-08-07T22:27:02.267336Z","submitted_at":"2025-11-01T03:29:56Z","title":"Efficiency vs. Alignment: Investigating Safety and Fairness Risks in Parameter-Efficient Fine-Tuning of LLMs","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T06:54:11.503514Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2511.00382"},"observation_digest":"sha256:736710a0bd7e4d9631b28569962cc77bee03a3737763a3f0bf0f1d3599f832eb","observation_id":"a3597325-4084-4808-804f-767418eaaee7","resolution":{"observed_at":"2026-08-04T06:54:11.503514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2602.08813","last_updated":"2026-05-12T17:22:35Z","snapshot_observed_at":"2026-08-06T12:22:42.069570Z","submitted_at":"2026-02-09T15:50:05Z","title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T05:33:42.965249Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2602.08813"},"observation_digest":"sha256:e708134f8b954f27c7464248ba237d905548702c4ee1b30421bf36f55ec046d7","observation_id":"30a2df69-b7c7-4aa0-bc5f-0db29504e29f","resolution":{"observed_at":"2026-05-16T05:37:24.200106Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.07754","last_updated":"2026-04-09T03:20:29Z","snapshot_observed_at":"2026-07-06T22:57:00.904627Z","submitted_at":"2026-04-09T03:20:29Z","title":"The Art of (Mis)alignment: How Fine-Tuning Methods Effectively Misalign and Realign LLMs in Post-Training","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-10T18:18:56.476698Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.07754"},"observation_digest":"sha256:b2bf7316c137c200990e67042e1e84cb2811060d19e40b0bba6fb35a1949358e","observation_id":"539c423a-598c-47bd-a544-9f2cdb64d3e4","resolution":{"observed_at":"2026-05-11T00:45:50.626971Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.17215","last_updated":"2026-04-19T02:52:33Z","snapshot_observed_at":"2026-08-07T13:05:16.901779Z","submitted_at":"2026-04-19T02:52:33Z","title":"Continual Safety Alignment via Gradient-Based Sample Selection","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T07:16:53.472918Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.17215"},"observation_digest":"sha256:c551810c4c096cd9f70f22b074e64b29f8ea3cf58359cb22e955b9119febcd05","observation_id":"4a20819c-bd6d-4dac-97c7-d940b96d1f11","resolution":{"observed_at":"2026-05-10T07:16:54.723710Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.17396","last_updated":"2026-04-19T11:59:58Z","snapshot_observed_at":"2026-08-01T18:23:32.699442Z","submitted_at":"2026-04-19T11:59:58Z","title":"Representation-Guided Parameter-Efficient LLM Unlearning","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-10T06:01:46.885030Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.17396"},"observation_digest":"sha256:5f5f569901b76f13eacbb4b56623819a4ae8968600ee52eafc3fbe4b8e2db762","observation_id":"3982e68c-4587-4aa8-b7ef-d5b29f6f0bb2","resolution":{"observed_at":"2026-05-10T06:06:19.395425Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.23338","last_updated":"2026-05-06T17:17:02Z","snapshot_observed_at":"2026-08-06T19:46:39.219000Z","submitted_at":"2026-04-25T14:57:15Z","title":"A Systematic Survey of Security Threats and Defenses in LLM-Based AI Agents: A Layered Attack Surface Framework","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-08T07:53:13.746141Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.23338"},"observation_digest":"sha256:e39a1b88258a1a3fe580ffc62cabefdffa7e089b18540f14d0de47f7ac8bb17b","observation_id":"ac357e15-b452-4ffa-9014-d6ea0e5def20","resolution":{"observed_at":"2026-05-11T20:51:09.500341Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2604.24902","last_updated":"2026-04-27T18:34:08Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T18:34:08Z","title":"Safety Drift After Fine-Tuning: Evidence from High-Stakes Domains","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-07T17:53:57.169962Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2604.24902"},"observation_digest":"sha256:4934f2eac88a3f44f9d5c5c0727d03fb504bd404c75cc9093223eb49e51ab44e","observation_id":"6c007f23-3612-4364-9134-03e5d47b5c66","resolution":{"observed_at":"2026-05-11T23:16:14.005373Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.02914","last_updated":"2026-04-08T05:27:33Z","snapshot_observed_at":"2026-08-02T18:20:53.616601Z","submitted_at":"2026-04-08T05:27:33Z","title":"When Safety Geometry Collapses: Fine-Tuning Vulnerabilities in Agentic Guard Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T18:43:12.298529Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.02914"},"observation_digest":"sha256:33cae8e5cdc0c3afaaa1b803bd005f62587080868334b93e30e82cda44664e5e","observation_id":"026f3de0-b29c-4319-8c5b-8b46131ae44a","resolution":{"observed_at":"2026-05-11T00:05:49.280147Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.10998","last_updated":"2026-05-09T15:52:29Z","snapshot_observed_at":"2026-08-03T01:46:31.309604Z","submitted_at":"2026-05-09T15:52:29Z","title":"Few-Shot Truly Benign DPO Attack for Jailbreaking LLMs","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-13T07:06:46.387088Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.10998"},"observation_digest":"sha256:c18217c0a22cc79581cdbf43827894428d6a431c0f5dadf2b38446cd21484d90","observation_id":"6b05f4bd-9405-42b9-bc68-ec7f5a25c467","resolution":{"observed_at":"2026-05-13T07:07:27.049638Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.12705","last_updated":"2026-05-12T20:08:00Z","snapshot_observed_at":"2026-08-06T21:05:32.361086Z","submitted_at":"2026-05-12T20:08:00Z","title":"Early Data Exposure Improves Robustness to Subsequent Fine-Tuning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T20:45:27.673290Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.12705"},"observation_digest":"sha256:f4f7b3dfc1ac013eac6532a15729b8c46d397043b42543916a9a71e1f366866b","observation_id":"ca6d5dbd-88cc-414b-a563-b525b536752d","resolution":{"observed_at":"2026-05-14T20:47:58.535416Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.14605","last_updated":"2026-05-24T08:34:13Z","snapshot_observed_at":"2026-07-06T23:25:58.895612Z","submitted_at":"2026-05-14T09:22:14Z","title":"One Step to the Side: Why Defenses Against Malicious Finetuning Fail Under Adaptive Adversaries","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-30T21:01:25.549340Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.14605"},"observation_digest":"sha256:dd92c9113f00209064f16a64d3933be33983c127ae302ff9106ea89492db71b7","observation_id":"363a98b2-587f-40db-bea0-4c105b6ccaa9","resolution":{"observed_at":"2026-06-30T21:05:04.065497Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.16471","last_updated":"2026-05-15T13:53:02Z","snapshot_observed_at":"2026-07-06T23:27:38.955917Z","submitted_at":"2026-05-15T13:53:02Z","title":"From AI-Generated Content to Agentic Action: Security and Safety Threats in Generative AI","version":1},"reference_index":145,"source":"pdf_text","source_observed_at":"2026-05-20T18:08:24.901025Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.16471"},"observation_digest":"sha256:82ff361c820da258a29ba87d01c6b8a28f0c4b665e5023603d4691cb7449198a","observation_id":"f2e55edf-6079-4b2a-bbfc-0111bee5acf3","resolution":{"observed_at":"2026-05-20T18:08:50.464701Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.21674","last_updated":"2026-05-20T19:31:07Z","snapshot_observed_at":"2026-07-06T23:32:05.693503Z","submitted_at":"2026-05-20T19:31:07Z","title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-22T09:35:51.862736Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.21674"},"observation_digest":"sha256:d24e61728b15968188c6320386f97028b8f7918b4fc8644302ca106749b56cb4","observation_id":"0b84fdc6-85e8-4316-85e1-129d7639b7ab","resolution":{"observed_at":"2026-05-22T09:36:21.003990Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.24154","last_updated":"2026-05-22T19:22:17Z","snapshot_observed_at":"2026-07-06T23:34:13.859813Z","submitted_at":"2026-05-22T19:22:17Z","title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T16:03:12.728352Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.24154"},"observation_digest":"sha256:11584edfed60548822a38d39462fae6905552285bd48499de52c2e8af675f954","observation_id":"edb54687-7a19-4cea-a398-f22302efe7ad","resolution":{"observed_at":"2026-06-30T16:04:52.548154Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.26526","last_updated":"2026-05-26T04:18:42Z","snapshot_observed_at":"2026-08-03T14:37:23.753557Z","submitted_at":"2026-05-26T04:18:42Z","title":"Open-Weight LLM Fine-Tuning Defenses are Susceptible to Simple Attacks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T19:23:47.574214Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.26526"},"observation_digest":"sha256:2264cf9d0bb055afb6a0501aaef649154638bd03127869daf0f51e2279f09e18","observation_id":"7cacd102-5732-4aed-8bc9-feaef39b7a1c","resolution":{"observed_at":"2026-06-29T19:23:53.646667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.28030","last_updated":"2026-05-27T06:36:22Z","snapshot_observed_at":"2026-08-07T03:38:26.044946Z","submitted_at":"2026-05-27T06:36:22Z","title":"SPARD: Defending Harmful Fine-Tuning Attack via Safety Projection with Relevance-Diversity Data Selection","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T13:49:56.311711Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.28030"},"observation_digest":"sha256:cd8e022d72b4a8ae8c392165f7c86495758677f035f4289784a2a9a76fb56f85","observation_id":"42583a00-abd0-4214-a5eb-7b6dfc4c5c72","resolution":{"observed_at":"2026-06-29T13:53:28.660121Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.28896","last_updated":"2026-05-27T11:54:23Z","snapshot_observed_at":"2026-07-06T23:38:23.512460Z","submitted_at":"2026-05-27T11:54:23Z","title":"Feature Geometry of LoRA Adapters: A Sparse Autoencoder Analysis of Representational Divergence in Fine-Tuned Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T13:48:36.304776Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.28896"},"observation_digest":"sha256:bed76d09204826b26272e971bc1e1fef0fea26e1297ebcda3536d9079d4d1297","observation_id":"86a45327-cef7-4f8f-bcae-13b1650f33df","resolution":{"observed_at":"2026-06-29T13:53:28.777900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.29396","last_updated":"2026-05-28T05:46:38Z","snapshot_observed_at":"2026-08-02T05:05:10.834203Z","submitted_at":"2026-05-28T05:46:38Z","title":"Aligned but Fragile: Enhancing LLM Safety Robustness via Zeroth-Order Optimization","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T07:41:03.219581Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.29396"},"observation_digest":"sha256:a1726bc2807a159a14a2fd98fcf0b1df668d9e95fd3793882aabf6a614a00198","observation_id":"d0ffbb66-00a2-4a1e-8735-f2177e4e3f6b","resolution":{"observed_at":"2026-06-29T07:43:13.494789Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2605.30640","last_updated":"2026-05-28T22:48:54Z","snapshot_observed_at":"2026-08-07T18:30:05.985217Z","submitted_at":"2026-05-28T22:48:54Z","title":"CSULoRA: Closest Safe Update Low-Rank Adaptation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T08:25:01.052975Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2605.30640"},"observation_digest":"sha256:b001e33ec173055fefc5e63720c695c8cba28414e0f69ea6d3af7471a169589d","observation_id":"bf2da3a5-6d73-4855-ad5b-43ac28e24aad","resolution":{"observed_at":"2026-06-29T08:33:15.752828Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.00160","last_updated":"2026-05-29T09:04:27Z","snapshot_observed_at":"2026-08-02T07:42:30.512493Z","submitted_at":"2026-05-29T09:04:27Z","title":"DataShield: Safety-degrading Data Filtering for LLM Benign Instruction Fine-Tuning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T22:09:02.498712Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.00160"},"observation_digest":"sha256:b14084f763dbc9fca43b5411ddf373c1d6f1e6602fbd743161a995e01b6ba62f","observation_id":"916ab45f-0e62-42bf-937a-b2dac39d344a","resolution":{"observed_at":"2026-07-01T19:46:10.220470Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.01695","last_updated":"2026-06-01T05:01:01Z","snapshot_observed_at":"2026-08-07T06:22:33.028268Z","submitted_at":"2026-06-01T05:01:01Z","title":"CANARY: Zero-Label Detection of Fine-Tuning Contamination in Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:01.257829Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.01695"},"observation_digest":"sha256:10c8206c02c334aaa07d9a847ae6ad0591beb14ec853a394a6e5c4a2ea1f9036","observation_id":"c1a7d728-736d-4eb8-b6dc-091baa690fe3","resolution":{"observed_at":"2026-07-01T22:26:18.023666Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.02111","last_updated":"2026-06-01T11:43:53Z","snapshot_observed_at":"2026-08-02T15:25:06.321153Z","submitted_at":"2026-06-01T11:43:53Z","title":"Jailbreaking Multimodal Large Language Models using Multi-Clip Video","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T15:16:48.957645Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.02111"},"observation_digest":"sha256:907839687c655f604f77eb29b503db4c84b227abecfe170cd35f2a2644177717","observation_id":"0d5f89a3-ee3c-4a53-9222-020f6c0cd61a","resolution":{"observed_at":"2026-07-01T22:36:17.122384Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.07631","last_updated":"2026-05-31T04:28:21Z","snapshot_observed_at":"2026-08-07T01:48:42.889654Z","submitted_at":"2026-05-31T04:28:21Z","title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T17:37:51.505359Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.07631"},"observation_digest":"sha256:4e8e975d6274ee8cc99cba49c2eda8ae12f45cef04f82bbcd6d237b65790d45a","observation_id":"f7687dd7-8ec9-48f2-ad93-d68d17261cf4","resolution":{"observed_at":"2026-07-01T20:56:13.889072Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.11316","last_updated":"2026-06-09T18:01:19Z","snapshot_observed_at":"2026-08-01T20:26:14.151904Z","submitted_at":"2026-06-09T18:01:19Z","title":"Sch\\\"utzen: Evaluating LLM Safety in Bulgarian and German Contexts","version":1},"reference_index":132,"source":"arxiv_source","source_observed_at":"2026-06-27T13:32:18.368158Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.11316"},"observation_digest":"sha256:2dd0af7584138faa1e8c49cc9408d44cfd35c48f1bd16f08957e841d5575fc0c","observation_id":"d5d76cf4-66d7-428d-b646-4fc822d144a0","resolution":{"observed_at":"2026-07-03T04:57:38.392850Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.12342","last_updated":"2026-06-10T17:15:28Z","snapshot_observed_at":"2026-07-06T23:51:17.719874Z","submitted_at":"2026-06-10T17:15:28Z","title":"ALIGNBEAM : Inference-Time Alignment Transfer via Cross-Vocabulary Logit Mixing","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T10:02:21.293918Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.12342"},"observation_digest":"sha256:16727abf582f1fb8f43b341f536005fc25dc402e9e742cb2f5471e76257c584a","observation_id":"6b09bb8b-3d4c-4d57-b297-7c818bb3df68","resolution":{"observed_at":"2026-07-03T10:27:56.364793Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.15980","last_updated":"2026-06-18T23:56:29Z","snapshot_observed_at":"2026-07-06T23:52:37.922109Z","submitted_at":"2026-06-14T19:07:22Z","title":"Do Activation Monitors Survive Model Updates? Benchmarking, Predicting, and Repairing Activation-Monitor Staleness","version":2},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-06-27T03:24:24.714121Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.15980"},"observation_digest":"sha256:7faae5c7bbd95674b70ea802bf6a76fa437f38e465a746a9381654b0eb7080b1","observation_id":"75df2934-0219-4bd2-aefc-b5a00ad87e95","resolution":{"observed_at":"2026-07-03T17:58:47.686325Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.19168","last_updated":"2026-06-17T15:11:43Z","snapshot_observed_at":"2026-08-05T23:22:13.413659Z","submitted_at":"2026-06-17T15:11:43Z","title":"Beyond Safe Data: Pretraining-Stage Alignment with Regular Safety Reflection","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-06-26T20:40:15.506976Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.19168"},"observation_digest":"sha256:3eb1f9a68d4ac1b7a39931d372ed807fb0ed36450232c9690d25728c663a9da4","observation_id":"0be25dfd-52ef-48ac-a9fa-a35c7f8ac810","resolution":{"observed_at":"2026-07-04T01:09:18.893870Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.22676","last_updated":"2026-06-21T21:30:15Z","snapshot_observed_at":"2026-08-07T16:50:09.374879Z","submitted_at":"2026-06-21T21:30:15Z","title":"Skin-Deep: A Geometric Diagnostic for Alignment Fragility in Large Language Model Representations","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-26T10:23:47.981377Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.22676"},"observation_digest":"sha256:bc6786144b99f1c9c174505e1c5e2d12d856e8d41d12a6b181c2323f4f267918","observation_id":"ff6e584a-c301-4845-b460-02cd23924670","resolution":{"observed_at":"2026-07-04T09:09:43.661267Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.30263","last_updated":"2026-06-29T13:11:49Z","snapshot_observed_at":"2026-08-05T03:55:13.444958Z","submitted_at":"2026-06-29T13:11:49Z","title":"Defending Against Harmful Supervision Hidden in Benign Samples","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T05:29:04.802064Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.30263"},"observation_digest":"sha256:7c14784de1c5968d2a47698e73f8499129f388bdf490ac25707d64c31db66145","observation_id":"376d516e-062b-4eae-b132-ebc3db2dad25","resolution":{"observed_at":"2026-06-30T14:24:45.321547Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":"2310.02949","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-04T09:09:43.659817Z","title":"Y ., Zhao, X., and Lin, D","venue":null,"work_id":"c19e9353-3a5c-4f79-9b83-3a458ebf7c01","year":2023},"citing_paper":{"arxiv_id":"2606.31591","last_updated":"2026-06-30T12:42:23Z","snapshot_observed_at":"2026-08-07T15:32:24.940758Z","submitted_at":"2026-06-30T12:42:23Z","title":"Evil Spectra: How Optimisers can Amplify or Suppress Emergent Misalignment","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-07-01T06:20:11.322710Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2606.31591"},"observation_digest":"sha256:c0ea77793370e53423dab16983a99a1091fc437d38dc967b62e7de36f09a3ec6","observation_id":"45e0734a-62f9-480b-9689-95425a50aea7","resolution":{"observed_at":"2026-07-01T09:45:39.623496Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-07-12T07:33:43.015966Z","title":"Shadow alignment: The ease of subverting safely-aligned language models.arXiv preprint arXiv:2310.02949,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02714","last_updated":"2026-07-07T12:39:55Z","snapshot_observed_at":"2026-08-07T02:57:17.862073Z","submitted_at":"2026-07-02T19:05:07Z","title":"Not All Refusals Are Equal: How Safety Alignment Fails Cybersecurity at Scale","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T07:33:43.015966Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2607.02714"},"observation_digest":"sha256:48f59297facfffd09fc2619eeb667ff88cc53fee37356d9b804bf2a1809468dd","observation_id":"a23f879c-557c-4abf-a2e4-8c4812be1ddb","resolution":{"observed_at":"2026-07-12T07:33:43.015966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-02T10:00:10.311613Z","title":"Shadow alignment: The ease of subverting safely-aligned language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16242","last_updated":"2026-06-26T03:29:59Z","snapshot_observed_at":"2026-08-06T19:59:54.568684Z","submitted_at":"2026-06-26T03:29:59Z","title":"TRACE: Trajectory-Based Safety Patch Learning for LLM Post-Training Realignment","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T10:00:10.311613Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2607.16242"},"observation_digest":"sha256:4f70d7653999cb0cacf56b0c8c4a8d30b6757476aa499609a8e4ad6b32c26c6c","observation_id":"6271f3ad-1a30-4338-bb64-853b857090af","resolution":{"observed_at":"2026-08-02T10:00:10.311613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02949","snapshot_observed_at":"2026-08-02T07:44:12.284016Z","title":"Shadow alignment: The ease of subverting safely-aligned language models.arXiv preprint arXiv:2310.02949,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22676","last_updated":"2026-07-10T11:11:48Z","snapshot_observed_at":"2026-08-06T09:33:49.866018Z","submitted_at":"2026-07-10T11:11:48Z","title":"How LLM Task-Adaptation Reshapes Alignment: A Multi-dimensional Study of Behavioral and Representational Drift","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T07:44:12.284016Z"},"links":{"cited_paper":"/paper/2310.02949","citing_paper":"/paper/2607.22676"},"observation_digest":"sha256:caa7c3dcd1a7248287b7e8e5c9bc0d1d4431dcadb82992aeeab81445ed3c31cf","observation_id":"6dc0aa6b-640a-4bd9-8fbc-554c13c04373","resolution":{"observed_at":"2026-08-02T07:44:12.284016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.02949/citation-record","integrity":"/paper/2310.02949/integrity","json":"/paper/2310.02949/citation-record.json","paper":"/paper/2310.02949"},"outbound":[],"paper":{"arxiv_id":"2310.02949","last_updated":"2023-10-04T16:39:31Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T11:20:01.991656Z","submitted_at":"2023-10-04T16:39:31Z","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 48 inbound Pith citation observations for arXiv:2310.02949."}