{"as_of":"2026-08-06T15:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a6f2eca1f8faacdf362f9b2d023973e9b2a51c693359f2d3f0a963e4b682868d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":45,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:48:01.382809Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":4,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2409.18169","last_updated":"2026-04-23T18:48:49Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:55:22Z","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","version":6},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-05-23T20:58:16.237327Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2409.18169"},"observation_digest":"sha256:66cac5eb9f3e759e553f1981b7e2112bd07124c464744192ab4031a0dc7a0eb1","observation_id":"18e07020-4633-4cca-ac77-ddc13a509810","resolution":{"observed_at":"2026-05-23T20:58:26.408949Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2503.02574","last_updated":"2026-05-18T17:54:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-04T12:55:07Z","title":"LLM-Safety Evaluations Lack Robustness","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-23T01:26:45.402983Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2503.02574"},"observation_digest":"sha256:27819b9c3ce8cd7c62313c48f28d42399bd56dc363b3bb0fc7f2beb09fe1b921","observation_id":"0a3315e6-5034-416a-a3a2-574c9ae1d9e1","resolution":{"observed_at":"2026-05-23T01:27:21.321187Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2506.05171","last_updated":"2026-04-08T13:36:47Z","snapshot_observed_at":"2026-07-06T21:37:23.562748Z","submitted_at":"2025-06-05T15:46:25Z","title":"Towards provable probabilistic safety for scalable embodied AI systems","version":3},"reference_index":138,"source":"pdf_text","source_observed_at":"2026-05-19T11:00:27.799347Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2506.05171"},"observation_digest":"sha256:861ed2fffaf9ec98cfcaed04608ac6d6f1ca2986beafe6f4cc7b5f8fce874180","observation_id":"69257a21-2ff3-458d-9a67-ac2adfc991f8","resolution":{"observed_at":"2026-05-19T11:02:15.176084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-06T15:48:01.382809Z","title":"Safety alignment should be made more than just a few tokens deep","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14987","last_updated":"2025-07-20T14:47:03Z","snapshot_observed_at":"2026-08-06T15:41:01.200040Z","submitted_at":"2025-07-20T14:47:03Z","title":"AlphaAlign: Incentivizing Safety Alignment with Extremely Simplified Reinforcement Learning","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T15:48:01.382809Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2507.14987"},"observation_digest":"sha256:d368a3e4dff97a4a1877e5f1c5507e807cf8cdd21f56c761476eae08439d897a","observation_id":"c4b5728f-b8ff-4436-83de-53d40cfe133f","resolution":{"observed_at":"2026-08-06T15:48:01.382809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2508.04204","last_updated":"2026-05-06T06:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-06T08:35:10Z","title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-19T01:02:07.088724Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2508.04204"},"observation_digest":"sha256:6e45c26be080a3f61365eb2f31af26102b37bef9916d34039db6772407a19e67","observation_id":"31a3c0fc-9898-45f0-8b3b-81da89cdccf5","resolution":{"observed_at":"2026-05-19T01:02:54.865061Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-06T11:51:41.439991Z","title":", Panda, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.09994","last_updated":"2025-09-08T03:30:40Z","snapshot_observed_at":"2026-08-06T11:51:19.809847Z","submitted_at":"2025-07-30T02:02:58Z","title":"Whisper Smarter, not Harder: Adversarial Attack on Partial Suppression","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T11:51:41.439991Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2508.09994"},"observation_digest":"sha256:da264f2058fd57425f1a97d60e0dc2cd543bec9ebb377e207fd668068d77f280","observation_id":"63596195-b6c2-4022-9ea1-47f6072fc253","resolution":{"observed_at":"2026-08-06T11:51:41.439991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-04T09:47:24.461017Z","title":"Safety alignment should be made more than just a few tokens deep, 2024.URL https://arxiv","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.13698","last_updated":"2026-07-14T06:18:50Z","snapshot_observed_at":"2026-08-04T22:28:44.623658Z","submitted_at":"2025-10-15T15:57:17Z","title":"Attention Misses Visual Risk: Risk-Adaptive Steering for Multimodal Safety Alignment","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T09:47:24.461017Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2510.13698"},"observation_digest":"sha256:605383921d961d3e9d0a73d44248e51b0a7068485951206f39a9095da5c76011","observation_id":"b18694bb-f4ed-470d-95df-23f3ff646588","resolution":{"observed_at":"2026-08-04T09:47:24.461017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-04T07:07:54.996168Z","title":"Safety alignment should be made more than just a few tokens deep, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.04694","last_updated":"2026-07-01T16:09:28Z","snapshot_observed_at":"2026-08-04T07:07:51.334462Z","submitted_at":"2025-10-30T22:13:31Z","title":"Reasoning Up the Instruction Ladder for Controllable Language Models","version":5},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-04T07:07:54.996168Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2511.04694"},"observation_digest":"sha256:e37c1b6155cb0930474f6794e6002fc741e54a45bba2feb070deab057eebfac3","observation_id":"9a85dacc-f99a-4773-9cb3-851478a24a55","resolution":{"observed_at":"2026-08-04T07:07:54.996168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2512.10998","last_updated":"2025-12-10T17:25:55Z","snapshot_observed_at":"2026-07-06T22:38:45.907386Z","submitted_at":"2025-12-10T17:25:55Z","title":"SCOUT: A Defense Against Data Poisoning Attacks in Fine-Tuned Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T23:26:48.405593Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2512.10998"},"observation_digest":"sha256:010691373df04dfbb078056b8411b517189df054cc2a748d6422ce918b589a6b","observation_id":"cfbb6043-d7a0-4edb-9bf6-cf0e2bbfa64c","resolution":{"observed_at":"2026-05-16T23:28:40.713873Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2602.07340","last_updated":"2026-05-21T07:39:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-07T03:46:33Z","title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T11:17:03.104902Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2602.07340"},"observation_digest":"sha256:8afa3383bcf3764770464405ae4f1fc44ed7b9986f55f00c6bc43a62b425d84d","observation_id":"796e4efe-4123-4436-9ed9-85fff8302d09","resolution":{"observed_at":"2026-05-22T11:21:29.072310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2602.08813","last_updated":"2026-05-12T17:22:35Z","snapshot_observed_at":"2026-08-06T12:22:42.069570Z","submitted_at":"2026-02-09T15:50:05Z","title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T05:33:42.965249Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2602.08813"},"observation_digest":"sha256:59a60b2d501a3dc901b513e8a03f423c0670074d6783564b3c8a784e65767809","observation_id":"c534c6af-9e58-40e0-bbba-33d68b94a85d","resolution":{"observed_at":"2026-05-16T05:37:24.169526Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2604.07709","last_updated":"2026-06-03T21:15:24Z","snapshot_observed_at":"2026-07-13T00:19:28.134640Z","submitted_at":"2026-04-09T01:54:33Z","title":"IatroBench: Pre-Registered Evidence of Iatrogenic Harm from AI Safety Measures","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T18:25:53.037936Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2604.07709"},"observation_digest":"sha256:9d09c2cddf1d6b529b164367c22cfaa7b9349fe282360b4792252348247a2bb0","observation_id":"aefd7fba-2534-46b6-8ec3-068284866237","resolution":{"observed_at":"2026-05-11T00:35:52.662413Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-07-13T00:19:33.861692Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.07709","last_updated":"2026-06-03T21:15:24Z","snapshot_observed_at":"2026-07-13T00:19:28.134640Z","submitted_at":"2026-04-09T01:54:33Z","title":"IatroBench: Pre-Registered Evidence of Iatrogenic Harm from AI Safety Measures","version":4},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-07-13T00:19:33.861692Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2604.07709"},"observation_digest":"sha256:7a9313e458df52d0da20320203c2fb5555cf28122a3b055e1b9fbe2b259b6e7c","observation_id":"3731c79d-7bc6-4d28-9d1f-c74def9e77ad","resolution":{"observed_at":"2026-07-13T00:19:33.861692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2604.07831","last_updated":"2026-04-09T05:32:34Z","snapshot_observed_at":"2026-08-02T13:15:13.204183Z","submitted_at":"2026-04-09T05:32:34Z","title":"Are GUI Agents Focused Enough? Automated Distraction via Semantic-level UI Element Injection","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T18:29:27.373741Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2604.07831"},"observation_digest":"sha256:7a6bcde074dfb3b2e2c7f1d02cf951572dc80241090bf3fa3a91c47e158feb66","observation_id":"a2dfbc05-f563-47c7-aa30-380075ba452c","resolution":{"observed_at":"2026-05-11T00:30:52.978026Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2604.09544","last_updated":"2026-07-03T15:37:04Z","snapshot_observed_at":"2026-08-01T18:28:17.308608Z","submitted_at":"2026-04-10T17:58:31Z","title":"Large Language Models Generate Harmful Responses Using a Distinct Mechanism, Shared Across Harm Types","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T17:08:25.471462Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2604.09544"},"observation_digest":"sha256:4e8c0576cee1141ec64e9fab896dea4cefd011e2dccbc5cf05d5297da52cbc61","observation_id":"c548783a-0369-45e7-9738-d533958b6ae5","resolution":{"observed_at":"2026-05-11T07:31:00.406851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2604.10403","last_updated":"2026-04-12T01:37:45Z","snapshot_observed_at":"2026-07-06T22:59:01.129724Z","submitted_at":"2026-04-12T01:37:45Z","title":"Latent Instruction Representation Alignment: defending against jailbreaks, backdoors and undesired knowledge in LLMs","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T16:41:52.440793Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2604.10403"},"observation_digest":"sha256:12f32e7fcccdc5515da5bf7f33a00bef0afd5ae6e649e265bb9a9f4de929472b","observation_id":"5f398ca0-4b21-4259-9724-775fd701064c","resolution":{"observed_at":"2026-05-11T08:21:00.097938Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.01687","last_updated":"2026-05-03T02:55:30Z","snapshot_observed_at":"2026-07-06T23:14:52.417213Z","submitted_at":"2026-05-03T02:55:30Z","title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T16:00:32.413225Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.01687"},"observation_digest":"sha256:c266fe842bd64410177c5f753ef2779bed9039a9b87568805f4ab532b9ea74fa","observation_id":"ccc0fa3b-dfe9-4e8b-b4b9-e7df34205fd6","resolution":{"observed_at":"2026-05-11T09:31:00.975345Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.02946","last_updated":"2026-05-01T11:54:55Z","snapshot_observed_at":"2026-07-31T02:35:59.178763Z","submitted_at":"2026-05-01T11:54:55Z","title":"RouteHijack: Routing-Aware Attack on Mixture-of-Experts LLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-09T19:22:00.217729Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.02946"},"observation_digest":"sha256:d339a7dbd64e64d957d73f6a4869a66b9e1ee32936aa5c28315d7c716c492c21","observation_id":"a2869571-ad0f-4ee1-8546-bbc2319adc3a","resolution":{"observed_at":"2026-05-11T15:46:18.735027Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.08930","last_updated":"2026-05-09T13:05:00Z","snapshot_observed_at":"2026-07-06T23:21:06.773761Z","submitted_at":"2026-05-09T13:05:00Z","title":"Internalizing Safety Understanding in Large Reasoning Models via Verification","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T01:50:59.283409Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.08930"},"observation_digest":"sha256:71091df3269c945aa46ada4bb2c31fdd7066df4edc782a1f7417bb54278e17e7","observation_id":"af70cf46-cef4-42b9-b766-97bcd16061eb","resolution":{"observed_at":"2026-05-12T01:51:14.234952Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.12813","last_updated":"2026-05-31T17:51:51Z","snapshot_observed_at":"2026-07-06T23:24:27.821980Z","submitted_at":"2026-05-12T23:13:50Z","title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","version":1},"reference_index":191,"source":"arxiv_source","source_observed_at":"2026-05-14T20:13:10.814899Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.12813"},"observation_digest":"sha256:acfa9b2fc59213f41e47955cacb699c09494df4706d2ff735dffc2f325b9a0eb","observation_id":"f8c0cbb3-a741-458e-9e23-509d52f38403","resolution":{"observed_at":"2026-05-14T20:19:26.730208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.15239","last_updated":"2026-05-14T03:40:07Z","snapshot_observed_at":"2026-07-06T23:26:32.979566Z","submitted_at":"2026-05-14T03:40:07Z","title":"Reducing the Safety Tax in LLM Safety Alignment with On-Policy Self-Distillation","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-19T16:34:47.856606Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.15239"},"observation_digest":"sha256:b1b693717af01e6e764d026fdc152b51cc4d64466fd7df466f2c6a020724931b","observation_id":"b8910e1c-87c9-4d1a-ab7a-1ba54351b815","resolution":{"observed_at":"2026-05-19T16:37:39.886052Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.17413","last_updated":"2026-05-17T12:18:20Z","snapshot_observed_at":"2026-08-03T01:59:45.581393Z","submitted_at":"2026-05-17T12:18:20Z","title":"Ablating Safety: Mechanisms for Removing Alignment in Language Models for Security Applications","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-19T23:30:43.364230Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.17413"},"observation_digest":"sha256:db1bbb0c7684d392e0fde10e65e9da6e0e3d75cac3b76589db20d947a0eaba62","observation_id":"50336ac5-2974-406f-ac2d-16be30884918","resolution":{"observed_at":"2026-05-19T23:32:52.534733Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.20342","last_updated":"2026-05-21T10:36:34Z","snapshot_observed_at":"2026-07-06T23:30:58.549353Z","submitted_at":"2026-05-19T18:01:26Z","title":"ParaVT: Taming the Tool Prior Paradox for Parallel Tool Use in Agentic Video Reinforcement Learning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-21T07:32:12.180233Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.20342"},"observation_digest":"sha256:a9919f9852ab88afcad0ec208412098a1dc5c249581231dd6aa000a47070add8","observation_id":"32b70579-c676-4e14-9cb8-9813176209ca","resolution":{"observed_at":"2026-05-21T07:34:02.553168Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.20342","last_updated":"2026-05-21T10:36:34Z","snapshot_observed_at":"2026-07-06T23:30:58.549353Z","submitted_at":"2026-05-19T18:01:26Z","title":"ParaVT: Taming the Tool Prior Paradox for Parallel Tool Use in Agentic Video Reinforcement Learning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T08:59:28.405218Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.20342"},"observation_digest":"sha256:9784bbb2f65e663d17e5b0d30d7eb670690bed57297ee133365e3ef19368c3a6","observation_id":"0c86dbcc-5cd4-4e1c-85ed-99cd3007d537","resolution":{"observed_at":"2026-05-22T09:01:19.405588Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.20654","last_updated":"2026-06-03T08:00:52Z","snapshot_observed_at":"2026-08-02T18:56:31.274465Z","submitted_at":"2026-05-20T03:16:15Z","title":"REFLECTOR: Internalizing Step-wise Reflection against Indirect Jailbreak","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-21T06:16:01.040236Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.20654"},"observation_digest":"sha256:d403b26b3f55b2b28d4cda3a30a2d2114ab329ecca8c82df47493fdf70532945","observation_id":"45b8de59-d327-4b5c-a66e-34520b9d6e45","resolution":{"observed_at":"2026-05-21T06:19:41.984616Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2605.25603","last_updated":"2026-05-25T08:54:55Z","snapshot_observed_at":"2026-08-03T17:47:49.546009Z","submitted_at":"2026-05-25T08:54:55Z","title":"Detecting Unfaithful Chain-of-Thought via Circuit-Guided Internal-External Discrepancy","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T21:47:17.894881Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2605.25603"},"observation_digest":"sha256:7f0773ed2296114943c4f83f756bad3ea2656d81f122f347c4c66a7cb43f9359","observation_id":"e74004bf-4bd9-4474-990e-04ed45a96364","resolution":{"observed_at":"2026-06-29T21:53:59.549110Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.00651","last_updated":"2026-05-30T09:54:38Z","snapshot_observed_at":"2026-07-06T23:41:19.955034Z","submitted_at":"2026-05-30T09:54:38Z","title":"MESA: Improving MoE Safety Alignment via Decentralized Expertise","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-28T18:52:00.377915Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.00651"},"observation_digest":"sha256:60480e9019e10b7b510bca96e0effa71e73f264ae08e58b85a244a1cff0fc221","observation_id":"eeb1c9dd-8132-4ef8-9bf8-8e403c29f35d","resolution":{"observed_at":"2026-06-28T19:52:35.291030Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.01060","last_updated":"2026-06-07T07:03:09Z","snapshot_observed_at":"2026-08-02T21:46:05.814578Z","submitted_at":"2026-05-31T07:05:51Z","title":"MENTIS: What Belief Changes Under Alignment? Measuring Multi-Scale Latent Torsion in Language Models","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-28T17:23:52.388431Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.01060"},"observation_digest":"sha256:ed6d8ac79463adefa0dcd01c5fe8fc6cc2c443de15731096666809a94b8a0aa8","observation_id":"54d52469-bcff-45c6-be6b-ec2b969b7c84","resolution":{"observed_at":"2026-07-01T21:16:13.704818Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.04027","last_updated":"2026-06-01T18:10:21Z","snapshot_observed_at":"2026-07-06T23:44:14.261541Z","submitted_at":"2026-06-01T18:10:21Z","title":"MaskForge: Structure-Aware Adaptive Attacks for Jailbreaking Diffusion Large Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T13:43:51.171443Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.04027"},"observation_digest":"sha256:39a98f050814b51f182b8fa99d3119e4a6fd4cd1f0cc309987ddc5e89d88a08b","observation_id":"40e2d7f2-51b7-4498-9ae6-418d195d0f72","resolution":{"observed_at":"2026-07-01T23:56:24.322243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.08044","last_updated":"2026-08-04T12:50:00Z","snapshot_observed_at":"2026-08-06T15:35:07.716788Z","submitted_at":"2026-06-06T08:10:56Z","title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-27T20:04:17.744876Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.08044"},"observation_digest":"sha256:6d096f61ffb3af8fc3ba50d91ee69ba28a2c0436da9219401e7f61a75ebea4ca","observation_id":"d974c386-bb2b-494d-bd0a-1181b7910d2b","resolution":{"observed_at":"2026-07-02T20:57:23.042879Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.18284","last_updated":"2026-06-10T02:04:29Z","snapshot_observed_at":"2026-07-06T23:53:45.117607Z","submitted_at":"2026-06-10T02:04:29Z","title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-06-27T10:36:09.211639Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.18284"},"observation_digest":"sha256:af532762b28cfbbc113a32e4e527c132af1c3e07a4f1826c9c0f958a24178e74","observation_id":"db63a811-404a-41be-84c2-f94e3316018c","resolution":{"observed_at":"2026-07-03T08:57:48.136806Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.18673","last_updated":"2026-06-17T04:20:00Z","snapshot_observed_at":"2026-07-06T23:54:04.739534Z","submitted_at":"2026-06-17T04:20:00Z","title":"Understanding and Mitigating Prompt Leaking Attacks in Real-World LLM-Based Applications","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-26T20:47:36.337189Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.18673"},"observation_digest":"sha256:9011e9acf1d8f5d212fce3aca6ed240af7e8ef5644dc8ca2951dc85eef199f1c","observation_id":"3509ee17-abb0-4704-ad67-20241ed6968e","resolution":{"observed_at":"2026-07-04T00:59:20.668111Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.26515","last_updated":"2026-07-10T19:28:32Z","snapshot_observed_at":"2026-08-02T05:35:28.099281Z","submitted_at":"2026-06-25T01:40:10Z","title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-26T05:34:05.109347Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.26515"},"observation_digest":"sha256:b4c91c331c1018fe548ba1f5239cbb13aa36e93ea55427f2389b95292ab891ee","observation_id":"d4d13048-04c3-4378-bfaa-27f6348b12c1","resolution":{"observed_at":"2026-07-04T12:59:52.982067Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.26515","last_updated":"2026-07-10T19:28:32Z","snapshot_observed_at":"2026-08-02T05:35:28.099281Z","submitted_at":"2026-06-25T01:40:10Z","title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-30T10:22:49.075651Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.26515"},"observation_digest":"sha256:c5429a8503ce0e8957328a8d2ed0dd73cbe4514548e8d3bb0f32f9b1e97a10da","observation_id":"822981c2-9a40-4540-a0b8-1fb41754b31a","resolution":{"observed_at":"2026-06-30T11:54:38.971589Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-07-14T17:16:44.626166Z","title":"Safety alignment should be made more than just a few tokens deep.arXiv preprint arXiv:2406.05946, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.26515","last_updated":"2026-07-10T19:28:32Z","snapshot_observed_at":"2026-08-02T05:35:28.099281Z","submitted_at":"2026-06-25T01:40:10Z","title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-14T17:16:44.626166Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.26515"},"observation_digest":"sha256:875b75aa865b0f934244bb6ef596f56ca6b57d3496396243d730cb86dfec6b23","observation_id":"72b76032-4b80-44a6-9452-18ff921c8d81","resolution":{"observed_at":"2026-07-14T17:16:44.626166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2606.30263","last_updated":"2026-06-29T13:11:49Z","snapshot_observed_at":"2026-08-05T03:55:13.444958Z","submitted_at":"2026-06-29T13:11:49Z","title":"Defending Against Harmful Supervision Hidden in Benign Samples","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T05:29:04.802064Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2606.30263"},"observation_digest":"sha256:faf23e29d4c077606b8b3636a66eb287bdd4c3868f541f13ac74e437226e0581","observation_id":"08176c57-f6af-417e-bb42-155045e9ae68","resolution":{"observed_at":"2026-06-30T14:24:45.328038Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":"2406.05946","doi":"10.48550/arxiv.2406.05946","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Safety alignment should be made more than just a few tokens deep","venue":"arXiv (Cornell University)","work_id":"40539ea6-b6ba-4195-971b-53c52efa9597","year":2024},"citing_paper":{"arxiv_id":"2607.01239","last_updated":"2026-05-01T18:57:03Z","snapshot_observed_at":"2026-07-07T00:06:49.412663Z","submitted_at":"2026-05-01T18:57:03Z","title":"Breaking Safety at the Token Boundary: How BPE Tokenization Creates Exploitable Gaps in LLM Alignment","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-07-04T01:17:44.209685Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.01239"},"observation_digest":"sha256:ea0c58de7dcadf564b66062a28e7688d4377f195ab3dfc4094b79878b48429e8","observation_id":"52b6147b-d107-47f2-99c0-493177db0293","resolution":{"observed_at":"2026-07-04T01:19:20.146959Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-07-11T18:20:40.287006Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04510","last_updated":"2026-08-03T19:51:37Z","snapshot_observed_at":"2026-08-06T15:30:43.119303Z","submitted_at":"2026-07-05T21:23:15Z","title":"Transplanting, inverting, and preventing a misalignment persona: method-conditional emergent misalignment in Qwen2.5","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-11T18:20:40.287006Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.04510"},"observation_digest":"sha256:dbe29966c44ea75b3e7f1a885141c4976224ae262498ddeb15341a2fb8499eca","observation_id":"13b80043-0285-4561-8a97-cda8b8ec98cb","resolution":{"observed_at":"2026-07-11T18:20:40.287006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-07-11T12:36:24.747752Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04846","last_updated":"2026-07-06T09:18:35Z","snapshot_observed_at":"2026-08-05T20:29:56.293262Z","submitted_at":"2026-07-06T09:18:35Z","title":"Pretraining Curricula Enable Selective Fine-tuning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-11T12:36:24.747752Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.04846"},"observation_digest":"sha256:31ad0189339fed7ed67c7adc712a3193f5b4f999cf478d8601c8e0415da6ce5f","observation_id":"8b0e5569-2443-405d-8e81-3555d29cf70a","resolution":{"observed_at":"2026-07-11T12:36:24.747752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-07-13T00:42:24.432562Z","title":"arXiv preprint arXiv:2406.05946 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09053","last_updated":"2026-07-10T02:50:21Z","snapshot_observed_at":"2026-08-06T01:57:21.331702Z","submitted_at":"2026-07-10T02:50:21Z","title":"An Emergent Mirage: Is Emergent Misalignment and Realignment Indeed a Robust Phenomenon?","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-07-13T00:42:24.432562Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.09053"},"observation_digest":"sha256:f19237647472b8edb26f901f38385110fec714ff73547e6215c3401315000378","observation_id":"3df96879-e39b-4505-8a87-b1cbe822d585","resolution":{"observed_at":"2026-07-13T00:42:24.432562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-02T06:30:56.251349Z","title":"Safety alignment should be made more than just a few tokens deep","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14147","last_updated":"2026-07-14T09:30:34Z","snapshot_observed_at":"2026-08-04T23:11:22.577320Z","submitted_at":"2026-07-14T09:30:34Z","title":"Breaking Refusal in the First Half: A Mechanistic Study of the Prefill Jailbreak","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-02T06:30:56.251349Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.14147"},"observation_digest":"sha256:99ce1beba0da6233416a40f096204e20900649f525d0c48f7d52c7438f20a2a1","observation_id":"54e89284-a98d-4c11-8add-40007f5d5d99","resolution":{"observed_at":"2026-08-02T06:30:56.251349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-02T10:01:58.569910Z","title":"Safety alignment should be made more than just a few tokens deep.arXiv preprint arXiv:2406.05946,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16240","last_updated":"2026-06-26T03:03:15Z","snapshot_observed_at":"2026-08-04T11:34:03.080529Z","submitted_at":"2026-06-26T03:03:15Z","title":"Normalized Rewards for Preference Optimization","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T10:01:58.569910Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.16240"},"observation_digest":"sha256:f87c773f204204a1283f6898e5a89e8872617ec45de8bbe07ad86c428948caef","observation_id":"4c5b5df1-bb7b-4950-8fc5-89572e115e3b","resolution":{"observed_at":"2026-08-02T10:01:58.569910Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-01T18:54:57.776258Z","title":"arXiv preprint arXiv:2406.05946 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17152","last_updated":"2026-07-19T09:18:35Z","snapshot_observed_at":"2026-08-05T14:31:41.750601Z","submitted_at":"2026-07-19T09:18:35Z","title":"How Jailbreak Attacks Inform Safety Alignment: A Defender-Centric, Shapley-Based Evaluation of Jailbreak Contributions","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-01T18:54:57.776258Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.17152"},"observation_digest":"sha256:adf36ae756a9cfbce1755d133c6e5da4ad8a8a30f156364bb6e771052815cc5b","observation_id":"e94f6a3c-d03e-4e58-8f17-d7a4b741ab86","resolution":{"observed_at":"2026-08-01T18:54:57.776258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-02T10:20:54.536465Z","title":"arXiv preprint arXiv:2406.05946 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22643","last_updated":"2026-06-24T02:40:49Z","snapshot_observed_at":"2026-08-04T13:06:48.299222Z","submitted_at":"2026-06-24T02:40:49Z","title":"Reason Before You Retrieve: Agentic Planning for Multi-modal RAG","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-02T10:20:54.536465Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2607.22643"},"observation_digest":"sha256:a512f4a975c5b88cf5668762f722bde4d64a148fa46227ccdbabfe3c9427115c","observation_id":"18bb393f-b765-4321-a81a-61f84477da3b","resolution":{"observed_at":"2026-08-02T10:20:54.536465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-06T10:46:10.753746Z","title":"arXiv preprint arXiv:2406.05946 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05045","last_updated":"2026-08-05T16:55:59Z","snapshot_observed_at":"2026-08-06T15:52:15.470600Z","submitted_at":"2026-08-05T16:55:59Z","title":"Gradient Immunity: Null-Space Resistance to Malicious Fine-Tuning","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T10:46:10.753746Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2608.05045"},"observation_digest":"sha256:6dd7879ffc729cd35ec2e2c98a8ac9529ae7963098a8f93149f926627e958350","observation_id":"3f8ab28d-1a8c-44a7-b8ce-684b4d2bc84d","resolution":{"observed_at":"2026-08-06T10:46:10.753746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.05946/citation-record","integrity":"/paper/2406.05946/integrity","json":"/paper/2406.05946/citation-record.json","paper":"/paper/2406.05946"},"outbound":[],"paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-07-06T18:27:54.379381Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 45 inbound Pith citation observations for arXiv:2406.05946."}