{"as_of":"2026-08-18T20:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3e67a675c78542aa8270f4e1d95d38990f2fcb21437f9b9bf7bb982cb7032b25","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":71,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:18:52.479377Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-12T21:50:34.629915Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08302","last_updated":"2025-09-11T10:17:06Z","snapshot_observed_at":"2026-08-16T17:38:13.606752Z","submitted_at":"2024-11-13T02:45:21Z","title":"RED: Unleashing Token-Level Rewards from Holistic Feedback via Reward Redistribution","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-12T21:50:34.629915Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2411.08302"},"observation_digest":"sha256:a3be058b2fb6bbc67bf0c2e8b7dfefda129d0ca07ddb81f3bd942905c52d18dc","observation_id":"027b6d57-d2fe-414a-90a1-03ebb53350df","resolution":{"observed_at":"2026-08-12T21:50:34.629915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-12T05:56:46.428134Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.19939","last_updated":"2025-05-17T15:14:14Z","snapshot_observed_at":"2026-08-17T19:34:58.995686Z","submitted_at":"2024-11-29T18:56:37Z","title":"VLSBench: Unveiling Visual Leakage in Multimodal Safety","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-12T05:56:46.428134Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2411.19939"},"observation_digest":"sha256:107a9e8288fe8494018940737f76906add08594850fc476dbcfa96718f9fb42f","observation_id":"542ffbf9-5bab-40b6-94f0-a5710bb8e41d","resolution":{"observed_at":"2026-08-12T05:56:46.428134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-11T22:49:58.514761Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human pref- erence","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.03123","last_updated":"2025-06-16T21:12:38Z","snapshot_observed_at":"2026-08-16T14:55:41.338120Z","submitted_at":"2024-12-04T08:43:12Z","title":"Robust Multi-bit Text Watermark with LLM-based Paraphrasers","version":2},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-11T22:49:58.514761Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2412.03123"},"observation_digest":"sha256:04304b1b32c550fa5619fbaff03209baf3ea633a545b24dba5a782937b430d0a","observation_id":"59280202-38f0-494f-b3db-ed4b0f5e2bdb","resolution":{"observed_at":"2026-08-11T22:49:58.514761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-08-17T10:18:05.650518Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-05-11T23:08:34.312466Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2412.05579"},"observation_digest":"sha256:f81279b6c627ded7ed76a19c0d96caa55380fb7d4107e11b232091d9db157e0f","observation_id":"c07fd98b-a7fa-4a78-bc62-d6a2491ba213","resolution":{"observed_at":"2026-05-11T23:08:36.843119Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-11T16:14:25.014603Z","title":"Jiaming Ji, Donghai Hong, Borong Zhang, Boyuan Chen, Josef Dai, Boren Zheng, Tianyi Qiu, Boxun Li, and Yaodong Yang","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10257","last_updated":"2024-12-16T14:54:00Z","snapshot_observed_at":"2026-08-15T04:46:36.661032Z","submitted_at":"2024-12-13T16:26:34Z","title":"Targeted Angular Reversal of Weights (TARS) for Knowledge Removal in Large Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T16:14:25.014603Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2412.10257"},"observation_digest":"sha256:e04bd9f37fb40070f5b1d3044487ed6ab34f1da9b4e5bf63402fb79447aedee8","observation_id":"02230ff2-b5e2-494d-aec5-e82f8bc79f83","resolution":{"observed_at":"2026-08-11T16:14:25.014603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-11T11:09:13.150282Z","title":"Pku-saferlhf: A safety alignment pref- erence dataset for llama family models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.15838","last_updated":"2024-12-30T07:27:58Z","snapshot_observed_at":"2026-08-16T14:52:33.599355Z","submitted_at":"2024-12-20T12:27:16Z","title":"Align Anything: Training All-Modality Models to Follow Instructions with Language Feedback","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T11:09:13.150282Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2412.15838"},"observation_digest":"sha256:a2784e216e88dbb25addac718e9c02e5da4fb10561bfd0a228ecf88edcbc9d8c","observation_id":"b22ca459-5270-4b9e-9092-fe28562c7324","resolution":{"observed_at":"2026-08-11T11:09:13.150282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-10T23:27:37.742460Z","title":"Pku-saferlhf: A safety alignment preference dataset for llama family models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20412","last_updated":"2025-01-04T13:27:04Z","snapshot_observed_at":"2026-08-15T08:09:26.172599Z","submitted_at":"2024-12-29T09:35:56Z","title":"Multi-Objective Large Language Model Unlearning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T23:27:37.742460Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2412.20412"},"observation_digest":"sha256:80200937c4f5d7920079a4fe64cd7d0a46b067a86671aaf0e7bb892762245106","observation_id":"1c2e0654-df77-4799-972b-b57bcfff8d00","resolution":{"observed_at":"2026-08-10T23:27:37.742460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-10T21:17:48.381225Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.05336","last_updated":"2025-01-09T16:02:51Z","snapshot_observed_at":"2026-08-15T11:21:12.805156Z","submitted_at":"2025-01-09T16:02:51Z","title":"Stream Aligner: Efficient Sentence-Level Alignment via Distribution Induction","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T21:17:48.381225Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2501.05336"},"observation_digest":"sha256:72ee4c49c6ae1ef89e9c49422b6d5e7024b24af70becb71614d31554be8a6212","observation_id":"7447808b-12ec-4dbe-8aa2-aa4544fb5fcf","resolution":{"observed_at":"2026-08-10T21:17:48.381225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-10T18:53:50.107993Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.10945","last_updated":"2025-08-06T06:47:04Z","snapshot_observed_at":"2026-08-15T00:46:03.529876Z","submitted_at":"2025-01-19T04:56:55Z","title":"Gradient-Based Multi-Objective Deep Learning: Algorithms, Theories, Applications, and Beyond","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T18:53:50.107993Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2501.10945"},"observation_digest":"sha256:ffedbe2f86216c084e7c18ada39ac67b0ffdf99115283bb238e26d449c4c1f40","observation_id":"fa50e317-59f1-41fc-bd43-4e29d0173a98","resolution":{"observed_at":"2026-08-10T18:53:50.107993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-10T14:25:13.146914Z","title":"PKU-SafeRLHF: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.15453","last_updated":"2025-01-28T06:35:32Z","snapshot_observed_at":"2026-08-15T19:34:30.287297Z","submitted_at":"2025-01-26T08:49:46Z","title":"Data-adaptive Safety Rules for Training Reward Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T14:25:13.146914Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2501.15453"},"observation_digest":"sha256:6a27544726acac1db5387e29be17d610195bbbc94a61b508df4a9bd92ea593cc","observation_id":"b12fa28a-d167-4fcf-aeb5-fa8ad8b9afc4","resolution":{"observed_at":"2026-08-10T14:25:13.146914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-09T16:18:40.813673Z","title":"Pku-saferlhf: A safety alignment preference dataset for llama family models.arXiv preprint arXiv:2406.15513, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01208","last_updated":"2025-06-20T10:54:05Z","snapshot_observed_at":"2026-08-16T10:10:26.213081Z","submitted_at":"2025-02-03T09:59:32Z","title":"On Almost Surely Safe Alignment of Large Language Models at Inference-Time","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-09T16:18:40.813673Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2502.01208"},"observation_digest":"sha256:bde667773c2112a16f9fe90997222f2531545faebd97f8d53b336c1e3c5a70f0","observation_id":"e12b652f-af52-4be7-b387-036e5cfe24d3","resolution":{"observed_at":"2026-08-09T16:18:40.813673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-09T13:14:34.037586Z","title":"Pku-saferlhf: A safety alignment preference dataset for llama family models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02153","last_updated":"2025-02-04T09:31:54Z","snapshot_observed_at":"2026-08-16T16:34:47.536642Z","submitted_at":"2025-02-04T09:31:54Z","title":"Vulnerability Mitigation for Safety-Aligned Language Models via Debiasing","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-09T13:14:34.037586Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2502.02153"},"observation_digest":"sha256:a088026d219be782baa17da970216a091b791ad4db45bb7b191cf29abd80bffd","observation_id":"44ab9950-d943-44e9-9e48-8df82a5721f4","resolution":{"observed_at":"2026-08-09T13:14:34.037586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-08T23:50:35.767760Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04040","last_updated":"2025-05-30T09:43:42Z","snapshot_observed_at":"2026-08-15T09:31:20.952358Z","submitted_at":"2025-02-06T13:01:44Z","title":"Safety Reasoning with Guidelines","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T23:50:35.767760Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2502.04040"},"observation_digest":"sha256:0102a9e01e521fa7c9ef6e4ec097a3c6635f617d4322b1cbe8489210d97a13c1","observation_id":"53647d4d-ed38-470f-a8bb-4640aef25a66","resolution":{"observed_at":"2026-08-08T23:50:35.767760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2502.06387","last_updated":"2026-04-07T06:33:34Z","snapshot_observed_at":"2026-08-16T10:45:56.107224Z","submitted_at":"2025-02-10T12:15:27Z","title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-23T03:56:18.703995Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2502.06387"},"observation_digest":"sha256:484be12cab06c7f72707abe4d9c1aeb180a662981a002d6386ac454fdf641a9a","observation_id":"238801f1-b654-41a6-a352-bebadbb6a97d","resolution":{"observed_at":"2026-05-23T03:57:29.606200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-08T16:14:57.329365Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10441","last_updated":"2025-02-10T09:19:52Z","snapshot_observed_at":"2026-08-15T16:33:10.713147Z","submitted_at":"2025-02-10T09:19:52Z","title":"AI Alignment at Your Discretion","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-08T16:14:57.329365Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2502.10441"},"observation_digest":"sha256:56b938bc59104eb22dd9a0ee1dc695037e5b3aebf0ce29d9afc0bc481884e038","observation_id":"d882c0ce-848e-41dd-9599-3728f085f572","resolution":{"observed_at":"2026-08-08T16:14:57.329365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-16T12:18:52.479377Z","title":", Hong, D","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.13120","last_updated":"2025-04-29T14:51:47Z","snapshot_observed_at":"2026-08-16T12:12:11.843693Z","submitted_at":"2025-04-17T17:38:18Z","title":"Probing and Inducing Combinational Creativity in Vision-Language Models","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-16T12:18:52.479377Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2504.13120"},"observation_digest":"sha256:79b86d3ab72f93b6b2979d671799da9e770f1185e84fdcaa74dac269fea4d63f","observation_id":"95db99d9-5da6-4d58-94bd-453783b2ad5a","resolution":{"observed_at":"2026-08-16T12:18:52.479377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-16T11:24:12.809889Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.15585","last_updated":"2025-06-09T02:36:20Z","snapshot_observed_at":"2026-08-17T21:55:02.668814Z","submitted_at":"2025-04-22T05:02:49Z","title":"A Comprehensive Survey in LLM(-Agent) Full Stack Safety: Data, Training and Deployment","version":4},"reference_index":265,"source":"pdf_text","source_observed_at":"2026-08-16T11:24:12.809889Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2504.15585"},"observation_digest":"sha256:561854f8b7b9ba2c73bfeedcaed18d3240160aa98705c4d48ddeba213f105d2e","observation_id":"d8c0ae58-70ef-46b8-a18d-15456ea83bfe","resolution":{"observed_at":"2026-08-16T11:24:12.809889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-15T23:52:01.523681Z","title":"PKU-SafeRLHF: Towards multi- level safety alignment for LLMs with human preference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06274","last_updated":"2025-05-06T15:42:31Z","snapshot_observed_at":"2026-08-18T19:03:01.287649Z","submitted_at":"2025-05-06T15:42:31Z","title":"PARM: Multi-Objective Test-Time Alignment via Preference-Aware Autoregressive Reward Model","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T23:52:01.523681Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.06274"},"observation_digest":"sha256:f3a98b534a5953b3300812c10e992f93f6d45cdfad3a5ccf890b2497df058675","observation_id":"2f93b59b-02a3-4d92-a9e1-683d7951790e","resolution":{"observed_at":"2026-08-15T23:52:01.523681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-15T20:50:10.534659Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.11875","last_updated":"2025-05-17T06:58:42Z","snapshot_observed_at":"2026-08-18T20:01:50.393122Z","submitted_at":"2025-05-17T06:58:42Z","title":"J1: Exploring Simple Test-Time Scaling for LLM-as-a-Judge","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-15T20:50:10.534659Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.11875"},"observation_digest":"sha256:13db8a95fab6603fa9bae9d43472883dc3ac13d1bb09778842f91f832ef976e9","observation_id":"06262224-13c3-4f23-8a03-21b71a1463e5","resolution":{"observed_at":"2026-08-15T20:50:10.534659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T15:12:26.903646Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16003","last_updated":"2025-05-21T20:40:30Z","snapshot_observed_at":"2026-08-18T13:34:52.447448Z","submitted_at":"2025-05-21T20:40:30Z","title":"SLMEval: Entropy-Based Calibration for Human-Aligned Evaluation of Large Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T15:12:26.903646Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.16003"},"observation_digest":"sha256:c6d6ab71d972686b261c8dc90fe3788a5e11bcaa22ca86b9fde2261b54586c14","observation_id":"2de9dde4-b320-4095-864b-2187a14f885d","resolution":{"observed_at":"2026-08-07T15:12:26.903646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:57:27.885966Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16869","last_updated":"2025-05-22T16:24:51Z","snapshot_observed_at":"2026-08-17T10:55:42.169363Z","submitted_at":"2025-05-22T16:24:51Z","title":"MPO: Multilingual Safety Alignment via Reward Gap Optimization","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T14:57:27.885966Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.16869"},"observation_digest":"sha256:35b5246533c5a0edb9efd7f913134c8afe5940569c52588e9a3c292913b614a3","observation_id":"795a9b5c-9ae4-401d-9fa3-edf03b080874","resolution":{"observed_at":"2026-08-07T14:57:27.885966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T15:09:58.475342Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17147","last_updated":"2025-05-22T08:22:57Z","snapshot_observed_at":"2026-08-18T11:00:28.155222Z","submitted_at":"2025-05-22T08:22:57Z","title":"MTSA: Multi-turn Safety Alignment for LLMs through Multi-round Red-teaming","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T15:09:58.475342Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.17147"},"observation_digest":"sha256:afc5f7060ca05c4dfd730d8e1d9c7e3cf3e8abdd6995de52e6e927c558b32224","observation_id":"e349f15f-85a8-4845-a1ec-db80e8b5f095","resolution":{"observed_at":"2026-08-07T15:09:58.475342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:52:28.092591Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17332","last_updated":"2025-05-22T22:56:58Z","snapshot_observed_at":"2026-08-15T09:04:33.753491Z","submitted_at":"2025-05-22T22:56:58Z","title":"SweEval: Do LLMs Really Swear? A Safety Benchmark for Testing Limits for Enterprise Use","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T14:52:28.092591Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.17332"},"observation_digest":"sha256:6a161bc7149f0bf65bb7631100fbab779d4ab7c83bd8e0fc6b02bbf188bc3cb4","observation_id":"fe891c49-9c68-4b27-a31d-636d3040cd64","resolution":{"observed_at":"2026-08-07T14:52:28.092591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2505.19134","last_updated":"2026-04-13T23:58:15Z","snapshot_observed_at":"2026-08-11T09:23:02.105788Z","submitted_at":"2025-05-25T13:11:55Z","title":"Incentivizing High-Quality Human Annotations with Golden Questions","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-19T13:41:26.730528Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.19134"},"observation_digest":"sha256:021d856b567c22a1ca9c04888639f8e9027fd2e175d8f438ff07ba928a468c3d","observation_id":"6390b21b-79ca-4333-85d3-e69482b152e3","resolution":{"observed_at":"2026-05-19T13:42:19.389017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:13:52.457692Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19690","last_updated":"2025-05-26T08:49:19Z","snapshot_observed_at":"2026-08-17T01:37:19.231497Z","submitted_at":"2025-05-26T08:49:19Z","title":"Beyond Safe Answers: A Benchmark for Evaluating True Risk Awareness in Large Reasoning Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:13:52.457692Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.19690"},"observation_digest":"sha256:a9a9decdee60f318d6b2001aab278d296bb84a52fb6958e3015e3b850ae37e30","observation_id":"62fb4ef6-d5c8-4425-9bab-d36d4cfe9f14","resolution":{"observed_at":"2026-08-07T14:13:52.457692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:11:31.054317Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19743","last_updated":"2025-08-16T11:40:47Z","snapshot_observed_at":"2026-08-17T20:47:52.242057Z","submitted_at":"2025-05-26T09:24:36Z","title":"Token-level Accept or Reject: A Micro Alignment Approach for Large Language Models","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:31.054317Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.19743"},"observation_digest":"sha256:b54239d9f1aacf6b280e929a532e98dad79a66c700b2995310612044c79120bc","observation_id":"5f05f867-ff0e-4065-8f01-bf7651d187a3","resolution":{"observed_at":"2026-08-07T14:11:31.054317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T14:00:05.888953Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20259","last_updated":"2025-05-26T17:40:40Z","snapshot_observed_at":"2026-08-10T01:20:47.147597Z","submitted_at":"2025-05-26T17:40:40Z","title":"Lifelong Safety Alignment for Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:00:05.888953Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.20259"},"observation_digest":"sha256:d831e7085c8fbcaa23d19f03a609cd4c5d55164f803bf2c3762a7f37b5c481b4","observation_id":"a158b6c6-fa85-4b43-9101-14ce2a31ae06","resolution":{"observed_at":"2026-08-07T14:00:05.888953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T12:44:20.174221Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.23729","last_updated":"2025-05-31T23:47:06Z","snapshot_observed_at":"2026-08-17T21:55:23.611352Z","submitted_at":"2025-05-29T17:56:05Z","title":"Bounded Rationality for LLMs: Satisficing Alignment at Inference-Time","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T12:44:20.174221Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2505.23729"},"observation_digest":"sha256:64c2755a16086b6ba8a0946b0ff43c1fd6dc1241c85e92b99509dca6915c251c","observation_id":"75620268-4462-4d24-b277-de0d52d9690e","resolution":{"observed_at":"2026-08-07T12:44:20.174221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T12:01:08.749164Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00782","last_updated":"2025-06-01T02:19:46Z","snapshot_observed_at":"2026-08-18T02:53:19.588193Z","submitted_at":"2025-06-01T02:19:46Z","title":"Jailbreak-R1: Exploring the Jailbreak Capabilities of LLMs via Reinforcement Learning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:01:08.749164Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2506.00782"},"observation_digest":"sha256:3d7b4e1ef1980752bf5b9fedb19ad5be17ab9bbef18cb71a0baa0a1900fb00e5","observation_id":"eefbd577-524a-4cc1-a8d0-3faa2c8147af","resolution":{"observed_at":"2026-08-07T12:01:08.749164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-07T11:04:06.823967Z","title":"Pku- saferlhf: Towards multi-level safety alignment for llms with human preference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03637","last_updated":"2025-07-07T09:53:22Z","snapshot_observed_at":"2026-08-08T03:23:19.289933Z","submitted_at":"2025-06-04T07:30:16Z","title":"RewardAnything: Generalizable Principle-Following Reward Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:04:06.823967Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2506.03637"},"observation_digest":"sha256:adb91e1d074d7d4ea796a03d5ad3b6fe41177c54e34cb59244b7edfd9493a7d8","observation_id":"eed00d0e-6f1d-4820-a85f-fc1adb3eada8","resolution":{"observed_at":"2026-08-07T11:04:06.823967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-15T19:32:11.801146Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16502","last_updated":"2025-06-19T17:56:16Z","snapshot_observed_at":"2026-08-17T18:24:07.102493Z","submitted_at":"2025-06-19T17:56:16Z","title":"Relic: Enhancing Reward Model Generalization for Low-Resource Indic Languages with Few-Shot Examples","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-15T19:32:11.801146Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2506.16502"},"observation_digest":"sha256:03759490e6ffcff74b9a187ca5c0b67eebff2da924609041599843e7047b9743","observation_id":"eda1e4c2-d2fb-449e-b7a7-bf446442c1ca","resolution":{"observed_at":"2026-08-15T19:32:11.801146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T22:08:22.809009Z","title":"arXiv:2406.15513","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22557","last_updated":"2025-08-13T10:28:17Z","snapshot_observed_at":"2026-08-14T00:37:15.582276Z","submitted_at":"2025-06-27T18:15:56Z","title":"MetaCipher: A Time-Persistent and Universal Multi-Agent Framework for Cipher-Based Jailbreak Attacks for LLMs","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:08:22.809009Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2506.22557"},"observation_digest":"sha256:d921e492c28bea571a6017fa33b31d46f1d5fb38c2a333f9bbcc325aaccebc51","observation_id":"dbe04a67-8c2d-4973-a02e-dd0938068495","resolution":{"observed_at":"2026-08-06T22:08:22.809009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T20:49:52.401761Z","title":"Jiaming Ji, Donghai Hong, Borong Zhang, Boyuan Chen, Josef Dai, Boren Zheng, Tianyi Qiu, Boxun Li, and Yaodong Yang","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01915","last_updated":"2025-07-02T17:25:26Z","snapshot_observed_at":"2026-08-14T04:11:54.990176Z","submitted_at":"2025-07-02T17:25:26Z","title":"Gradient-Adaptive Policy Optimization: Towards Multi-Objective Alignment of Large Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:49:52.401761Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2507.01915"},"observation_digest":"sha256:cfe3881a5e5c04b25db1971924df3a78e5911f396cc13a3691dcf48fd2debcc9","observation_id":"3f47598e-ec16-4990-aec8-02e509777422","resolution":{"observed_at":"2026-08-06T20:49:52.401761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T17:48:07.137146Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09973","last_updated":"2025-07-14T06:43:00Z","snapshot_observed_at":"2026-08-08T02:31:23.778902Z","submitted_at":"2025-07-14T06:43:00Z","title":"Tiny Reward Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T17:48:07.137146Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2507.09973"},"observation_digest":"sha256:6bb669388a7dc1f7ad3f0342cdda2e64695c569da96cd79bac12aed355df2f9b","observation_id":"42fe3188-5383-423d-b63f-299496c0b3a7","resolution":{"observed_at":"2026-08-06T17:48:07.137146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T17:38:16.221861Z","title":"& Yang, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11502","last_updated":"2025-07-14T15:09:05Z","snapshot_observed_at":"2026-08-12T23:31:52.879519Z","submitted_at":"2025-07-14T15:09:05Z","title":"HKGAI-V1: Towards Regional Sovereign Large Language Model for Hong Kong","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:38:16.221861Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2507.11502"},"observation_digest":"sha256:a7675698a3875a1cda48f534ca480b14f27b9c61e6feae35b1e72791002214c2","observation_id":"a1f836f5-fb40-4bdb-9479-ea1315d2ea15","resolution":{"observed_at":"2026-08-06T17:38:16.221861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T19:08:23.564199Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11544","last_updated":"2025-07-08T23:58:01Z","snapshot_observed_at":"2026-08-18T00:49:35.206839Z","submitted_at":"2025-07-08T23:58:01Z","title":"The Safety Gap Toolkit: Evaluating Hidden Dangers of Open-Source Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T19:08:23.564199Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2507.11544"},"observation_digest":"sha256:1a5a038ef801a8d2e5e4cbd6706fc19cd10b26350d0cc333d8f27765328bec5b","observation_id":"bd353d43-b1bd-435a-ae9f-85c017eed29b","resolution":{"observed_at":"2026-08-06T19:08:23.564199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T15:14:22.849329Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16534","last_updated":"2025-07-26T12:33:42Z","snapshot_observed_at":"2026-08-15T01:11:58.606591Z","submitted_at":"2025-07-22T12:44:38Z","title":"Frontier AI Risk Management Framework in Practice: A Risk Analysis Technical Report","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:22.849329Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2507.16534"},"observation_digest":"sha256:20cbda9bba3703c67f9839644790c48544b44aa63bcc3003bda962b88cba1223","observation_id":"7cee2df6-4214-4911-bf5f-530698648edc","resolution":{"observed_at":"2026-08-06T15:14:22.849329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-06T13:54:39.777864Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19894","last_updated":"2026-08-15T13:49:25Z","snapshot_observed_at":"2026-08-18T20:17:29.262727Z","submitted_at":"2025-07-26T09:49:57Z","title":"Generative Model Unlearning: A Survey through Target Events, Unlearning Operators, and Evaluation Protocols","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-06T13:54:39.777864Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2507.19894"},"observation_digest":"sha256:06b10043eb25c22bc5dabcd9fa957d4234af7eed86f49020dfb98c5bab3412b3","observation_id":"8a1e5778-804d-43e5-9f4f-b251132c5784","resolution":{"observed_at":"2026-08-06T13:54:39.777864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T15:19:28.700858Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.20038","last_updated":"2025-09-04T09:23:46Z","snapshot_observed_at":"2026-08-15T11:36:17.519073Z","submitted_at":"2025-08-27T16:44:03Z","title":"Forewarned is Forearmed: Pre-Synthesizing Jailbreak-like Instructions to Enhance LLM Safety Guardrail to Potential Attacks","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T15:19:28.700858Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2508.20038"},"observation_digest":"sha256:a29d59667661140b31f9fef03b126004ab8ff4677787e0efd5a4852c559cb04b","observation_id":"eff3e40d-afc2-4c4a-adf1-c989463ac2dc","resolution":{"observed_at":"2026-08-05T15:19:28.700858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T15:08:54.032086Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21101","last_updated":"2025-08-28T07:05:24Z","snapshot_observed_at":"2026-08-17T16:38:55.095576Z","submitted_at":"2025-08-28T07:05:24Z","title":"Beyond Prediction: Reinforcement Learning as the Defining Leap in Healthcare AI","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-05T15:08:54.032086Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2508.21101"},"observation_digest":"sha256:606fa96be7497cd189bf7cf1754665c81ccbbc1eae9bd7b8a2bf5de0fa35541b","observation_id":"c95821df-2fb0-4028-9f2c-4fb40de35681","resolution":{"observed_at":"2026-08-05T15:08:54.032086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2511.02623","last_updated":"2026-05-11T15:34:10Z","snapshot_observed_at":"2026-08-16T12:45:25.801851Z","submitted_at":"2025-11-04T14:52:58Z","title":"The Realignment Problem: When Right becomes Wrong in LLMs","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-18T01:28:29.056145Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2511.02623"},"observation_digest":"sha256:352e695b2227315bc28534490335324b37d23ecdba756eee3089e316331e92f0","observation_id":"223eb6c5-4b62-4bb9-b4aa-58793cd83de8","resolution":{"observed_at":"2026-05-18T01:30:35.748934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2512.10998","last_updated":"2025-12-10T17:25:55Z","snapshot_observed_at":"2026-08-13T10:29:50.028883Z","submitted_at":"2025-12-10T17:25:55Z","title":"SCOUT: A Defense Against Data Poisoning Attacks in Fine-Tuned Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T23:26:48.405593Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2512.10998"},"observation_digest":"sha256:e2f1fbd0be479a139e82b4d4ed3308cd27fa32a07d72738ffa262dcf4c450d6f","observation_id":"25368b2e-73a0-462d-b0d2-a24ebb1d63cc","resolution":{"observed_at":"2026-05-16T23:28:40.673267Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2602.07892","last_updated":"2026-05-12T03:21:50Z","snapshot_observed_at":"2026-08-11T15:42:35.247968Z","submitted_at":"2026-02-08T09:53:46Z","title":"Safety Alignment as Continual Learning: Mitigating the Alignment Tax via Orthogonal Gradient Projection","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T06:39:17.785715Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2602.07892"},"observation_digest":"sha256:910205e2608eaa6ee59212baa88962fefb99c7ec08c03d7a510009f46280d3f1","observation_id":"e537745e-1e07-4d3c-a064-b218c88a2be4","resolution":{"observed_at":"2026-05-16T06:40:42.297455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-07-13T10:41:51.251736Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference.arXiv preprint arXiv:2406.15513,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.04229","last_updated":"2026-04-05T19:08:51Z","snapshot_observed_at":"2026-08-14T15:31:53.970726Z","submitted_at":"2026-04-05T19:08:51Z","title":"Hierarchical Semantic Correlation-Aware Masked Autoencoder for Unsupervised Audio-Visual Representation Learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-13T10:41:51.251736Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.04229"},"observation_digest":"sha256:911e4d6af368e57a96e45c8cc1718a15dc0e8c58aca7355fdbbee08c7d669b40","observation_id":"be7a05a0-4339-4a04-b476-d341a4c317a7","resolution":{"observed_at":"2026-07-13T10:41:51.251736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2604.06833","last_updated":"2026-04-08T08:51:46Z","snapshot_observed_at":"2026-08-18T06:42:22.591745Z","submitted_at":"2026-04-08T08:51:46Z","title":"FedDetox: Robust Federated SLM Alignment via On-Device Data Sanitization","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T17:22:40.613937Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.06833"},"observation_digest":"sha256:a851a26dfa95dcc8438bcca1059affee76cbe5e921529d0f0c6bcb1bb2767728","observation_id":"4e1ba725-ec50-426f-8a97-8b3346a4580b","resolution":{"observed_at":"2026-05-11T06:56:01.706661Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2604.10673","last_updated":"2026-04-12T14:54:31Z","snapshot_observed_at":"2026-08-12T11:02:44.368523Z","submitted_at":"2026-04-12T14:54:31Z","title":"Principles Do Not Apply Themselves: A Hermeneutic Perspective on AI Alignment","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T15:23:33.330990Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.10673"},"observation_digest":"sha256:ef023d0fba4a6e1e5fba4e0588f3af591350c1c37a28abcf48084b60ef104121","observation_id":"36019f95-c876-420d-b347-bec1c4ad3951","resolution":{"observed_at":"2026-05-11T10:41:02.849441Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2604.17614","last_updated":"2026-04-19T20:58:25Z","snapshot_observed_at":"2026-08-14T08:41:23.060090Z","submitted_at":"2026-04-19T20:58:25Z","title":"Characterizing Model-Native Skills","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-10T05:42:49.694715Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.17614"},"observation_digest":"sha256:39e8e5d7135b6772cfca6a0e277bc960ebf64af053b4b6610e0a9bea32b6fc6a","observation_id":"f16c24b6-74c0-48d8-8c32-627d49b8987d","resolution":{"observed_at":"2026-05-10T05:56:11.648277Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-08-17T04:02:17.790778Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:4663e740b9a5892eea7d911c6db42576a18de778f0903ffb878776dd6dbba872","observation_id":"1116b0a4-0e5d-4143-9f40-f09da8eb3e57","resolution":{"observed_at":"2026-05-10T12:20:23.122827Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2604.19638","last_updated":"2026-04-21T16:27:20Z","snapshot_observed_at":"2026-08-11T13:56:53.724081Z","submitted_at":"2026-04-21T16:27:20Z","title":"SafetyALFRED: Evaluating Safety-Conscious Planning of Multimodal Large Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-10T02:21:29.463149Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2604.19638"},"observation_digest":"sha256:63ff430f30b4d64b3d793ee8c073988795f32f7fa85eca518f0ddf29fe13b712","observation_id":"a1430dbe-54c2-4812-8474-047c9a47b1cc","resolution":{"observed_at":"2026-05-10T02:22:20.648413Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.01899","last_updated":"2026-05-03T14:28:08Z","snapshot_observed_at":"2026-08-17T00:31:30.183586Z","submitted_at":"2026-05-03T14:28:08Z","title":"Disentangling Intent from Role: Adversarial Self-Play for Persona-Invariant Safety Alignment","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-09T17:24:54.796037Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.01899"},"observation_digest":"sha256:a2d88437561678f94fa311148828dbb8428aa1e39a250bf825804053c2004e83","observation_id":"421aee01-67c4-4e2e-9c11-22bc5ba68250","resolution":{"observed_at":"2026-05-11T16:21:07.450631Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.07105","last_updated":"2026-05-08T01:32:22Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T01:32:22Z","title":"Theoretical Limits of Language Model Alignment","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-11T01:18:37.614335Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.07105"},"observation_digest":"sha256:362aa249dded5cab496089ee154a79f17036fe7e9a7ed05da83ab38923421e44","observation_id":"443f9f7a-6264-41a6-b089-a718eeda634d","resolution":{"observed_at":"2026-05-11T04:30:57.130357Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.07982","last_updated":"2026-05-08T16:44:07Z","snapshot_observed_at":"2026-08-11T11:43:55.315312Z","submitted_at":"2026-05-08T16:44:07Z","title":"GLiGuard: Schema-Conditioned Classification for LLM Safeguard","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-11T03:19:11.495230Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.07982"},"observation_digest":"sha256:4d69ab79650a4a660c208f95cceec2c192f33fa6baec91ec85eb2e597eb73340","observation_id":"92d2c7c5-7c5f-4519-a9b3-d91e03622f41","resolution":{"observed_at":"2026-05-11T03:20:55.235357Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.09946","last_updated":"2026-05-12T21:07:09Z","snapshot_observed_at":"2026-08-17T12:34:53.939309Z","submitted_at":"2026-05-11T03:50:09Z","title":"Structure from Strategic Interaction & Uncertainty: Risk Sensitive Games for Robust Preference Learning","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-12T04:15:24.919355Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.09946"},"observation_digest":"sha256:b959c10611ea8cf63cdf041b254a93950213e5755cfa44cd94bc6edcd3509d41","observation_id":"d4cf753a-c288-435a-a064-76e6dee5dbb9","resolution":{"observed_at":"2026-05-12T06:26:26.493049Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.09946","last_updated":"2026-05-12T21:07:09Z","snapshot_observed_at":"2026-08-17T12:34:53.939309Z","submitted_at":"2026-05-11T03:50:09Z","title":"Structure from Strategic Interaction & Uncertainty: Risk Sensitive Games for Robust Preference Learning","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-14T22:03:05.102274Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.09946"},"observation_digest":"sha256:2e92a1dc545ea7971a1a8382b89b9fa010bf3ef51e96c25d6e2577cfd0f538a4","observation_id":"55a935d3-6ded-46ec-9a05-70d7f56a358a","resolution":{"observed_at":"2026-05-14T22:08:05.044304Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.11679","last_updated":"2026-05-13T09:28:34Z","snapshot_observed_at":"2026-08-17T09:51:33.434736Z","submitted_at":"2026-05-12T07:38:59Z","title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T01:03:10.263663Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.11679"},"observation_digest":"sha256:13ec1d0112acb6a38d50a0b0127e1d6b178a64c7486cfa3df4d91824cadc949c","observation_id":"14b214f3-f55b-43a4-8c5d-4fb4bae7ccc4","resolution":{"observed_at":"2026-05-13T01:07:00.573759Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.11679","last_updated":"2026-05-13T09:28:34Z","snapshot_observed_at":"2026-08-17T09:51:33.434736Z","submitted_at":"2026-05-12T07:38:59Z","title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-14T21:12:06.989077Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.11679"},"observation_digest":"sha256:5ceb3ebf121994010c340f0a786a03dc6c056661b2cf71d834ca9e9925d30f9d","observation_id":"79d9af60-23e5-4e2c-9a54-3f79909f8d3a","resolution":{"observed_at":"2026-05-14T21:12:58.932654Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.11712","last_updated":"2026-05-12T08:02:34Z","snapshot_observed_at":"2026-08-13T03:39:42.477344Z","submitted_at":"2026-05-12T08:02:34Z","title":"Toward Stable Value Alignment: Introducing Independent Modules for Consistent Value Guidance","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-13T07:03:14.415745Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.11712"},"observation_digest":"sha256:ea044d316bc72c3c19b97e4b3931ce5d29f1745dc85d0fc584f308e99af44112","observation_id":"78a7f548-801c-4d12-93df-971461538fbb","resolution":{"observed_at":"2026-05-13T07:07:27.645410Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.26315","last_updated":"2026-05-25T20:13:06Z","snapshot_observed_at":"2026-08-17T11:13:02.353611Z","submitted_at":"2026-05-25T20:13:06Z","title":"Curriculum Learning for Safety Alignment","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T22:25:31.739331Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.26315"},"observation_digest":"sha256:64f81b497b44fe0f2189cb1eecb33b464b59e6e925ed0935857df02f8276b2df","observation_id":"ee8bb0a1-922b-4ca0-b682-34ddbaad234b","resolution":{"observed_at":"2026-06-29T22:34:02.198090Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2605.29659","last_updated":"2026-05-28T09:21:42Z","snapshot_observed_at":"2026-08-12T14:52:35.122614Z","submitted_at":"2026-05-28T09:21:42Z","title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T09:11:58.843585Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2605.29659"},"observation_digest":"sha256:6e9e556647b8ff657f559f06c34a16e487906d65213d278f94356f9bf6288fa4","observation_id":"8ef90606-1dae-45e7-b924-a55ef5ee26a2","resolution":{"observed_at":"2026-06-29T09:13:15.988572Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2606.07335","last_updated":"2026-06-05T14:49:26Z","snapshot_observed_at":"2026-07-06T23:47:00.425715Z","submitted_at":"2026-06-05T14:49:26Z","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-06-27T21:55:48.561400Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2606.07335"},"observation_digest":"sha256:060c54fa24c3b3153f193c4c791a118d8b0e4bc5c7f7494903cf8edcf51bb1b5","observation_id":"97042b30-7a37-4aa3-a47b-2d98b0242395","resolution":{"observed_at":"2026-07-02T17:37:14.846238Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2606.07631","last_updated":"2026-05-31T04:28:21Z","snapshot_observed_at":"2026-08-07T01:48:42.889654Z","submitted_at":"2026-05-31T04:28:21Z","title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-28T17:37:51.505359Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2606.07631"},"observation_digest":"sha256:fc7548b55ab63caa0d51207250ea86df6cf068a6d1078784262c2fadcb408ad3","observation_id":"4358c7bf-53a8-46c3-bbff-052a33ebadd0","resolution":{"observed_at":"2026-07-01T20:56:13.867485Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2606.08044","last_updated":"2026-08-04T12:50:00Z","snapshot_observed_at":"2026-08-07T23:11:38.433528Z","submitted_at":"2026-06-06T08:10:56Z","title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-27T20:04:17.744876Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2606.08044"},"observation_digest":"sha256:c0330505a100fbdb56c8c116dcbaeae0451c2db959746871f1ef996b001aa4f4","observation_id":"0a43df0a-b0bd-4264-972b-f9e371709a1e","resolution":{"observed_at":"2026-07-02T20:57:23.032922Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-07-12T08:52:34.039089Z","title":"arXiv preprint arXiv:2406.15513 (2024) 16 H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.01392","last_updated":"2026-07-03T09:34:52Z","snapshot_observed_at":"2026-08-02T06:05:05.007648Z","submitted_at":"2026-07-01T18:50:14Z","title":"Multi-Objective Exploration and Preference Optimization via Mutual Information","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-12T08:52:34.039089Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.01392"},"observation_digest":"sha256:0ecb520e969ba6d80a467b13631a20500ddd081e1743840ff899d4d87b911d69","observation_id":"c595a4c3-30a6-4659-9233-e8b39098f680","resolution":{"observed_at":"2026-07-12T08:52:34.039089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2607.02047","last_updated":"2026-07-02T11:14:52Z","snapshot_observed_at":"2026-08-18T14:10:49.713931Z","submitted_at":"2026-07-02T11:14:52Z","title":"OpenSafeIntent: Evaluating Intent-Calibrated Safe Completion Across Dual-Use Prompt Sets","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-07-03T14:44:57.205766Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.02047"},"observation_digest":"sha256:dd2f25939ca005951870fcee4eaae53b616fb5079e716a178913c495588b0131","observation_id":"601de794-7e48-4b62-9180-effb850bbf3b","resolution":{"observed_at":"2026-07-03T14:48:32.496348Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":"2406.15513","doi":"10.48550/arxiv.2406.15513","metadata_source":"pith","pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference","venue":"cs.AI","work_id":"03df2169-4c1d-40f7-b5dc-edd5daaef02b","year":2024},"citing_paper":{"arxiv_id":"2607.07907","last_updated":"2026-07-08T20:42:46Z","snapshot_observed_at":"2026-08-14T01:59:45.772219Z","submitted_at":"2026-07-08T20:42:46Z","title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","version":1},"reference_index":276,"source":"arxiv_source","source_observed_at":"2026-07-10T15:38:58.361411Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.07907"},"observation_digest":"sha256:bddc79ad7b685c57055ae25b2554a7cdae08ea9923a87c9ca738eab4534b97d7","observation_id":"eb8b634a-2ac2-4c99-8aa1-543c4c1076f7","resolution":{"observed_at":"2026-07-10T15:47:23.388056Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-02T02:00:25.508169Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14485","last_updated":"2026-07-16T01:58:53Z","snapshot_observed_at":"2026-08-16T12:18:51.533898Z","submitted_at":"2026-07-16T01:58:53Z","title":"Step-Level Preference Learning for Generative Agents in Social Simulations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T02:00:25.508169Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.14485"},"observation_digest":"sha256:2ba99344b611236497cbfd9c94a36015e02f609ad3c335314f62a1af83f72ecf","observation_id":"dd406a54-ffe1-4442-ad2e-102896795bda","resolution":{"observed_at":"2026-08-02T02:00:25.508169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-02T10:00:10.523132Z","title":"Pku-saferlhf: Towards multi-level safety alignment for llms with human preference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16242","last_updated":"2026-06-26T03:29:59Z","snapshot_observed_at":"2026-08-16T04:57:30.898748Z","submitted_at":"2026-06-26T03:29:59Z","title":"TRACE: Trajectory-Based Safety Patch Learning for LLM Post-Training Realignment","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T10:00:10.523132Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.16242"},"observation_digest":"sha256:ef5be0da1ea8ea1228f914263ad3088e50d3573160c8192f26bd2898cf72e3cc","observation_id":"9d5a050a-24bd-4eac-ac8d-78d296a66c50","resolution":{"observed_at":"2026-08-02T10:00:10.523132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-02T13:36:55.305100Z","title":"doi:10.48550/arXiv.2406.15513 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21610","last_updated":"2026-05-20T04:43:02Z","snapshot_observed_at":"2026-08-12T15:36:12.578643Z","submitted_at":"2026-05-20T04:43:02Z","title":"SCOPE and SCION: A Benchmark and an Auditable Reference Pipeline for Schema Induction and Fusion from Text","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-02T13:36:55.305100Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.21610"},"observation_digest":"sha256:d7379a461ab6ede0469c7393203e576455b796c4eb028ecb925bf2f734dd2b17","observation_id":"aaeee69b-2017-43c0-a70b-38d6b7350112","resolution":{"observed_at":"2026-08-02T13:36:55.305100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-01T13:10:37.372979Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22716","last_updated":"2026-07-21T15:52:30Z","snapshot_observed_at":"2026-08-18T20:14:34.753850Z","submitted_at":"2026-07-21T15:52:30Z","title":"Visual Token Compression Enhances Robustness of MLLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T13:10:37.372979Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2607.22716"},"observation_digest":"sha256:426e38ed645da1bfd5978437ed0e97e78283f7d14e7f3ed297dcaa565155c5d1","observation_id":"c5a4afa0-c396-49aa-b748-5bb4d4808ac4","resolution":{"observed_at":"2026-08-01T13:10:37.372979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-11T16:35:24.771726Z","title":"arXiv preprint arXiv:2406.15513 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09490","last_updated":"2026-08-10T11:58:04Z","snapshot_observed_at":"2026-08-17T16:27:12.096112Z","submitted_at":"2026-08-10T11:58:04Z","title":"When Do Task Vectors Interfere? Mapping the Validity Boundaries of Weight-Space Composition","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T16:35:24.771726Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2608.09490"},"observation_digest":"sha256:9f3684a6b3470d2ab91fbb64f60c0f893d5e70bb6ee84050ae4c496d1432e013","observation_id":"354bbca7-6e30-44f8-8921-d118c8d1b3ca","resolution":{"observed_at":"2026-08-11T16:35:24.771726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15513","snapshot_observed_at":"2026-08-16T00:35:14.975581Z","title":"PKU-SafeRLHF: Towards multi-level safety alignment for llms with human preference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11797","last_updated":"2026-08-12T08:40:24Z","snapshot_observed_at":"2026-08-16T00:24:18.145611Z","submitted_at":"2026-08-12T08:40:24Z","title":"Orientation, not magnitude: the causal structure of task-vector interference in merged language models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-16T00:35:14.975581Z"},"links":{"cited_paper":"/paper/2406.15513","citing_paper":"/paper/2608.11797"},"observation_digest":"sha256:27cc3e9e8d34a279cf1d584ad5973a408a2d1585d3d5c88fac683349e3284b0b","observation_id":"86642dfe-8e5c-417f-a708-d95cbcb5edf8","resolution":{"observed_at":"2026-08-16T00:35:14.975581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.15513/citation-record","integrity":"/paper/2406.15513/integrity","json":"/paper/2406.15513/citation-record.json","paper":"/paper/2406.15513"},"outbound":[],"paper":{"arxiv_id":"2406.15513","last_updated":"2025-06-14T16:04:31Z","latest_version":3,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-16T13:40:57.893955Z","submitted_at":"2024-06-20T18:37:36Z","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 71 inbound Pith citation observations for arXiv:2406.15513."}