{"as_of":"2026-08-12T06:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:271cabc62658473fc5abc5087b4a90d4f20f2591a1e167f0ba2761f8f98e6e1b","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T11:50:26.030339Z","state":"measured"},{"denominator":81,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":81,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":66,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T22:57:01.733324Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":14,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2310.13548","last_updated":"2025-05-10T07:10:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-20T14:46:48Z","title":"Towards Understanding Sycophancy in Language Models","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T06:26:29.196349Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2310.13548"},"observation_digest":"sha256:1d9993d6f97e502530865206b4f9332b1b1f929c441425d4daf6b6028f9b18db","observation_id":"30866190-b7ec-43f6-9755-a980371398a7","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2403.07691","last_updated":"2024-03-14T07:47:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T14:34:08Z","title":"ORPO: Monolithic Preference Optimization without Reference Model","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-16T09:34:04.394588Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2403.07691"},"observation_digest":"sha256:14da090867de0b56a44193f9e55f757ca00778f048c358c587b0ba0100d88682","observation_id":"9051f36c-75a0-4c44-b076-337f21694f72","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-11T22:57:01.733324Z","title":"Understanding the effects of rlhf on llm generalisation and diversity, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02980","last_updated":"2024-12-09T22:23:41Z","snapshot_observed_at":"2026-08-12T03:24:59.349878Z","submitted_at":"2024-12-04T02:47:45Z","title":"Surveying the Effects of Quality, Diversity, and Complexity in Synthetic Data From Large Language Models","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-11T22:57:01.733324Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2412.02980"},"observation_digest":"sha256:43f099e2184b76aff20a307c314837bdfa5d1625a79af8c96e93233b1edf2f15","observation_id":"13209e60-0223-48ae-ad90-738155a0883a","resolution":{"observed_at":"2026-08-11T22:57:01.733324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-11T12:16:32.022520Z","title":"chosen” response and one of the remaining three at random as the “rejected","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.14516","last_updated":"2024-12-19T04:31:56Z","snapshot_observed_at":"2026-08-11T21:48:46.109109Z","submitted_at":"2024-12-19T04:31:56Z","title":"Cal-DPO: Calibrated Direct Preference Optimization for Language Model Alignment","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-11T12:16:32.022520Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2412.14516"},"observation_digest":"sha256:9c15287346e70f5e4aee2a9a7d861f6d7e06f3385956a72023719499a0279275","observation_id":"484e4191-a869-4b78-bbf9-af172e8cac7c","resolution":{"observed_at":"2026-08-11T12:16:32.022520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-10T22:49:13.413589Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00747","last_updated":"2025-01-01T06:33:45Z","snapshot_observed_at":"2026-08-11T11:10:02.175345Z","submitted_at":"2025-01-01T06:33:45Z","title":"DIVE: Diversified Iterative Self-Improvement","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T22:49:13.413589Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2501.00747"},"observation_digest":"sha256:f928784894efd2d8e47aade3ffafb86d0d07db61828f9960be8d4d29a59b2bf3","observation_id":"d404cf99-d334-4188-ac77-f64d431bc6ef","resolution":{"observed_at":"2026-08-10T22:49:13.413589Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-10T22:45:12.293458Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00911","last_updated":"2025-01-01T17:58:31Z","snapshot_observed_at":"2026-08-10T22:37:21.538654Z","submitted_at":"2025-01-01T17:58:31Z","title":"Aligning LLMs with Domain Invariant Reward Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T22:45:12.293458Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2501.00911"},"observation_digest":"sha256:44f9f3c02593c2a03f392dd90dbce521cbd9d07d656c390514309885e4a2347b","observation_id":"cac133d6-8dd6-4c6a-8397-d1f010ce696d","resolution":{"observed_at":"2026-08-10T22:45:12.293458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-10T22:27:31.290165Z","title":"arXiv preprint arXiv:2310.06452","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.01679","last_updated":"2025-01-03T07:47:59Z","snapshot_observed_at":"2026-08-10T22:20:16.405815Z","submitted_at":"2025-01-03T07:47:59Z","title":"Adaptive Few-shot Prompting for Machine Translation with Pre-trained Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:31.290165Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2501.01679"},"observation_digest":"sha256:98e5d156d096ce1f2e0520f43e50f8e988149bf168c696b8ca2dabe82feaadce","observation_id":"e67b475e-b965-4957-84be-5614ee3ffccf","resolution":{"observed_at":"2026-08-10T22:27:31.290165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T15:05:22.984029Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17149","last_updated":"2025-05-22T09:02:15Z","snapshot_observed_at":"2026-08-10T08:49:44.615981Z","submitted_at":"2025-05-22T09:02:15Z","title":"Large Language Models for Predictive Analysis: How Far Are They?","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T15:05:22.984029Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2505.17149"},"observation_digest":"sha256:0c4098584bd319ac0816f00545c6743042efd44b77578075d40ea8f4e7d92a15","observation_id":"8acb3879-ee10-48be-8a7c-e79a24b96458","resolution":{"observed_at":"2026-08-07T15:05:22.984029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T14:37:25.229173Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18447","last_updated":"2025-05-29T14:15:50Z","snapshot_observed_at":"2026-08-09T14:07:59.077153Z","submitted_at":"2025-05-24T01:09:10Z","title":"Pessimism Principle Can Be Effective: Towards a Framework for Zero-Shot Transfer Reinforcement Learning","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:37:25.229173Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2505.18447"},"observation_digest":"sha256:d38b8ab51734dc3eaf9d085f4ce16d6e5bf1057f5e6ea4c39a54d115689d23af","observation_id":"969b2ab7-23a2-4163-9c25-0a075435e67a","resolution":{"observed_at":"2026-08-07T14:37:25.229173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T14:19:03.201102Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19406","last_updated":"2025-05-26T01:42:38Z","snapshot_observed_at":"2026-08-11T08:26:46.992308Z","submitted_at":"2025-05-26T01:42:38Z","title":"Unveiling the Compositional Ability Gap in Vision-Language Reasoning Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:03.201102Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2505.19406"},"observation_digest":"sha256:6fa5e75a8a53b4dce673db69abcc697050b1db792b6eab5fb802e64cff2bcaba","observation_id":"c20c6894-9fcc-4cb0-a18d-f9177110cd8f","resolution":{"observed_at":"2026-08-07T14:19:03.201102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2505.23912","last_updated":"2026-05-13T21:21:09Z","snapshot_observed_at":"2026-08-11T16:17:43.473190Z","submitted_at":"2025-05-29T18:05:20Z","title":"LoVeC: Reinforcement Learning for Better Verbalized Confidence in Long-Form Generations","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-19T12:43:51.019983Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2505.23912"},"observation_digest":"sha256:d4b8d6c1724056312d4f058d4e6d73b059e861380d476025b1e662e3d5cf81d6","observation_id":"c0dcd5fb-12b7-40a6-b6b1-2f493cdcc838","resolution":{"observed_at":"2026-05-19T12:47:17.998837Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T11:16:25.248418Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02959","last_updated":"2025-06-03T14:52:44Z","snapshot_observed_at":"2026-08-10T16:52:56.034314Z","submitted_at":"2025-06-03T14:52:44Z","title":"HACo-Det: A Study Towards Fine-Grained Machine-Generated Text Detection under Human-AI Coauthoring","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:16:25.248418Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2506.02959"},"observation_digest":"sha256:a623a9b6560a8d0335c0ced4e88203bfcb7713f2b05da5b6be737a39ab40a192","observation_id":"ba295577-6be3-464f-8101-1f94facba3d2","resolution":{"observed_at":"2026-08-07T11:16:25.248418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T10:40:54.570349Z","title":"(2023, October)","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.04679","last_updated":"2025-06-05T06:57:28Z","snapshot_observed_at":"2026-08-07T20:31:12.431523Z","submitted_at":"2025-06-05T06:57:28Z","title":"Normative Conflicts and Shallow AI Alignment","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:40:54.570349Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2506.04679"},"observation_digest":"sha256:5defbb0bf3b2780377a8363802b1631f5ebec031608719213894470c8a4e1037","observation_id":"fd67e504-3c5c-4e2a-bf38-ee91c3a9c828","resolution":{"observed_at":"2026-08-07T10:40:54.570349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T05:03:13.573676Z","title":"Understanding the effects of rlhf on llm generalisation and diversity.arXiv preprint arXiv:2310.06452, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.08967","last_updated":"2025-06-13T10:07:42Z","snapshot_observed_at":"2026-08-08T15:39:54.287755Z","submitted_at":"2025-06-10T16:37:39Z","title":"Step-Audio-AQAA: a Fully End-to-End Expressive Large Audio Language Model","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T05:03:13.573676Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2506.08967"},"observation_digest":"sha256:3d4a12233999a85685a01625005a40539d58d946a6ad5e187dc61129e4aced7c","observation_id":"4784308c-b4a6-411a-855a-1dc0db1d4d6d","resolution":{"observed_at":"2026-08-07T05:03:13.573676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-07T04:29:51.203465Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10491","last_updated":"2025-09-01T09:21:41Z","snapshot_observed_at":"2026-08-10T16:20:27.900750Z","submitted_at":"2025-06-12T08:47:40Z","title":"Surface Fairness, Deep Bias: A Comparative Study of Bias in Language Models","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T04:29:51.203465Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2506.10491"},"observation_digest":"sha256:41d3b2bf67cca923d138e34d0fe53fc685ddd70c26b41f97c4037b2529d19070","observation_id":"cccd8799-141c-4517-9894-693752d3952a","resolution":{"observed_at":"2026-08-07T04:29:51.203465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-06T20:00:54.037131Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04072","last_updated":"2025-07-05T15:32:41Z","snapshot_observed_at":"2026-08-07T20:33:16.060921Z","submitted_at":"2025-07-05T15:32:41Z","title":"CTR-Guided Generative Query Suggestion in Conversational Search","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T20:00:54.037131Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2507.04072"},"observation_digest":"sha256:1bcccc16649757278a23dab49d3da1dedff7f42bd9eb6ed62561f26d1b6c21bc","observation_id":"6f35f197-7f74-4c8a-9b95-02d5b7baa79d","resolution":{"observed_at":"2026-08-06T20:00:54.037131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2507.06419","last_updated":"2026-06-04T20:44:16Z","snapshot_observed_at":"2026-08-11T09:29:02.629422Z","submitted_at":"2025-07-08T21:56:33Z","title":"Teach a Reward Model to Correct Itself: Reward Guided Adversarial Failure Discovery for Robust Reward Modeling","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-19T05:16:22.274580Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2507.06419"},"observation_digest":"sha256:242b1d0681f4d172cbe11c43d448797f928c18bc4f948fe0808873105193f32e","observation_id":"a2a64e26-8a74-474a-a8f0-5ee525036ded","resolution":{"observed_at":"2026-05-19T05:17:05.883933Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-06T15:46:27.773166Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15092","last_updated":"2025-07-20T19:14:43Z","snapshot_observed_at":"2026-08-09T04:23:52.892130Z","submitted_at":"2025-07-20T19:14:43Z","title":"A Penalty Goes a Long Way: Measuring Lexical Diversity in Synthetic Texts Under Prompt-Influenced Length Variations","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T15:46:27.773166Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2507.15092"},"observation_digest":"sha256:f193a7c820764b916ac82da6dccc0545cecf07cd50fbcb1f349ee0670e1d9995","observation_id":"c4763010-7a79-4ad8-ab8f-3ffa5754e0e0","resolution":{"observed_at":"2026-08-06T15:46:27.773166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2507.21934","last_updated":"2026-04-19T21:21:19Z","snapshot_observed_at":"2026-08-05T11:40:21.062589Z","submitted_at":"2025-07-29T15:48:12Z","title":"Culinary Crossroads: A RAG Framework for Enhancing Diversity in Cross-Cultural Recipe Adaptation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T02:20:03.766702Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2507.21934"},"observation_digest":"sha256:3637d6e100fd2e848d3a35d797317b118cf4f8fef7337810c71b46c52434e4ef","observation_id":"c2bd1add-cf10-4f58-9a09-4d93513f50ea","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-06T11:16:04.694338Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.22879","last_updated":"2025-07-31T16:54:43Z","snapshot_observed_at":"2026-08-06T11:16:02.241906Z","submitted_at":"2025-07-30T17:55:06Z","title":"RecGPT Technical Report","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T11:16:04.694338Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2507.22879"},"observation_digest":"sha256:7751ad948a2a59c5d30535b31c15a25efe8ebb2338dbe59a4a622c9b0437792d","observation_id":"1e80e756-06d2-4fc0-942c-2348326119eb","resolution":{"observed_at":"2026-08-06T11:16:04.694338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-06T05:29:13.238929Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.01781","last_updated":"2025-08-03T14:37:16Z","snapshot_observed_at":"2026-08-07T21:40:51.636920Z","submitted_at":"2025-08-03T14:37:16Z","title":"A comprehensive taxonomy of hallucinations in Large Language Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T05:29:13.238929Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2508.01781"},"observation_digest":"sha256:2e4c612485e58bcab2ecc50b153b25bbf7e9d5a726e8799d62aa67871f05e1b9","observation_id":"066160e2-6d6a-4cf5-9ba3-380cfa25d568","resolution":{"observed_at":"2026-08-06T05:29:13.238929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T23:06:49.596674Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.06026","last_updated":"2025-08-08T05:25:54Z","snapshot_observed_at":"2026-08-08T09:51:41.415322Z","submitted_at":"2025-08-08T05:25:54Z","title":"Temporal Self-Rewarding Language Models: Decoupling Chosen-Rejected via Past-Future","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T23:06:49.596674Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2508.06026"},"observation_digest":"sha256:201fd385bad66d5622d193a07296cabf30fd25e6184cbc1b695876beca20d017","observation_id":"31ba34a6-ad6a-4d61-9c6c-372f29c13ac3","resolution":{"observed_at":"2026-08-05T23:06:49.596674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2508.16771","last_updated":"2026-07-02T23:23:00Z","snapshot_observed_at":"2026-08-05T17:05:43.834611Z","submitted_at":"2025-08-22T20:08:09Z","title":"EyeMulator: Improving Code Language Models by Mimicking Human Visual Attention","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-18T20:54:30.449792Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2508.16771"},"observation_digest":"sha256:0601e2f3d1a281277f76a3e87562a0cea44fad2d78050f41485221953a2cec40","observation_id":"7b8e508f-0763-4ffd-a1b2-567c015f9161","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T16:19:07.651949Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18739","last_updated":"2025-08-26T07:11:44Z","snapshot_observed_at":"2026-08-11T08:00:14.005607Z","submitted_at":"2025-08-26T07:11:44Z","title":"Beyond Quality: Unlocking Diversity in Ad Headline Generation with Large Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T16:19:07.651949Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2508.18739"},"observation_digest":"sha256:e29c95283fcf2928b02827a0e86fdf2dd2354d3948e6064f5f39d6ec6050001b","observation_id":"3a26016f-4c03-4e91-8dcb-15e1f3edfaf4","resolution":{"observed_at":"2026-08-05T16:19:07.651949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T11:56:55.106254Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02170","last_updated":"2025-09-03T10:39:50Z","snapshot_observed_at":"2026-08-06T16:26:57.555793Z","submitted_at":"2025-09-02T10:22:46Z","title":"Avoidance Decoding for Diverse Multi-Branch Story Generation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T11:56:55.106254Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2509.02170"},"observation_digest":"sha256:62ba96ae27fd64b138cce290e07644462b81914e0799269055454f95998bd3de","observation_id":"9aaa1ec7-4b77-49e4-a0f2-a30c7cd3ee04","resolution":{"observed_at":"2026-08-05T11:56:55.106254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T22:59:14.507748Z","title":"Understanding the effects of rlhf on llm generalisation and diversity.arXiv preprint arXiv:2310.06452,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.06941","last_updated":"2025-09-08T17:52:56Z","snapshot_observed_at":"2026-08-07T12:11:53.268632Z","submitted_at":"2025-09-08T17:52:56Z","title":"Outcome-based Exploration for LLM Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T22:59:14.507748Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2509.06941"},"observation_digest":"sha256:e669b9b3f1c0d3528ab20adef9eb5a18933d7406c5ee60bd4ed70eec6b05380c","observation_id":"94662f91-9583-418f-89b4-06d067697620","resolution":{"observed_at":"2026-08-04T22:59:14.507748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T21:52:09.158688Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07706","last_updated":"2026-02-26T07:39:32Z","snapshot_observed_at":"2026-08-09T03:58:38.701437Z","submitted_at":"2025-09-09T13:10:49Z","title":"FHIR-RAG-MEDS: Integrating HL7 FHIR with Retrieval-Augmented Large Language Models for Enhanced Medical Decision Support","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-04T21:52:09.158688Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2509.07706"},"observation_digest":"sha256:119d720befc91002ebd935f446aa9baab558fb809d12273d086b451f54d9c9c1","observation_id":"d27768e6-cbb6-4fa1-b25f-ee3eb1833c0e","resolution":{"observed_at":"2026-08-04T21:52:09.158688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T20:24:40.594050Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08653","last_updated":"2025-09-11T11:18:17Z","snapshot_observed_at":"2026-08-07T20:32:11.024796Z","submitted_at":"2025-09-10T14:49:12Z","title":"Generative Data Refinement: Just Ask for Better Data","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-04T20:24:40.594050Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2509.08653"},"observation_digest":"sha256:5137ff0d4d4ed46c3d297cd259022247c84211347340545653d7bd2153e4af3d","observation_id":"53443ca6-d594-4faf-ac2f-91f04d57e61a","resolution":{"observed_at":"2026-08-04T20:24:40.594050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":255,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:2599fa29aaea65b4bc38fd3ec13c491939802de2426875c14ebd320130fc28e4","observation_id":"5cc7c9b8-dafb-4441-910b-8ef0388ba4b2","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T15:17:39.451092Z","title":"Understanding the effects of RLHF on LLM generalisation and diversity,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.20324","last_updated":"2026-06-04T17:26:56Z","snapshot_observed_at":"2026-08-07T20:32:31.507281Z","submitted_at":"2025-09-24T17:11:35Z","title":"RAG Security and Privacy: Formalizing the Threat Model and Attack Surface","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T15:17:39.451092Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2509.20324"},"observation_digest":"sha256:b350dc3ec07b2ec771ccd619d2a3a2feb9362e7bdaeda4df0b8acf3ad315fcc2","observation_id":"e6198dcb-079c-4f2b-9327-599327b11ed8","resolution":{"observed_at":"2026-08-04T15:17:39.451092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T07:21:35.867471Z","title":"Understanding the effects of RLHF on LLM generalisation and diversity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.26707","last_updated":"2026-07-15T04:49:02Z","snapshot_observed_at":"2026-08-07T05:29:04.731147Z","submitted_at":"2025-10-30T17:09:09Z","title":"Value Drifts: Tracing Value Alignment During LLM Post-Training","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-04T07:21:35.867471Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2510.26707"},"observation_digest":"sha256:62d45c9999fa4f4180d5095cf7453750d11ecedc9cfbd1747ecebfe1249a7ae8","observation_id":"e4f0ae8d-e82a-4240-b65c-68f0c77ce503","resolution":{"observed_at":"2026-08-04T07:21:35.867471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-03T23:27:36.198906Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.05933","last_updated":"2026-06-24T14:20:52Z","snapshot_observed_at":"2026-08-07T19:34:44.238702Z","submitted_at":"2025-11-08T08:56:29Z","title":"Reinforcement Learning Improves Traversal of Parametric Knowledge in LLMs","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-03T23:27:36.198906Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2511.05933"},"observation_digest":"sha256:064c6b9ee2882f5f5acc05057bd75f9ada2847f6224180c83da78ebede8d5025","observation_id":"2096b447-0bd7-44a4-9045-96fa54ce940b","resolution":{"observed_at":"2026-08-03T23:27:36.198906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-03T14:34:51.316303Z","title":"Akshay Krishnamurthy, Keegan Harris, Dylan J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.20111","last_updated":"2026-06-04T03:41:38Z","snapshot_observed_at":"2026-08-03T14:34:47.811827Z","submitted_at":"2025-12-23T07:11:26Z","title":"ABBEL: Learning Natural-Language Belief States for Memory-Efficient Interaction","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T14:34:51.316303Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2512.20111"},"observation_digest":"sha256:076781dba24fde5c90d65bd6bbd1bfaa88a0040d0fdab347fba3a2dd0f71c846","observation_id":"639f128c-b7ae-4ab1-9747-6b6c800ee5c7","resolution":{"observed_at":"2026-08-03T14:34:51.316303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2601.12538","last_updated":"2026-01-18T18:58:23Z","snapshot_observed_at":"2026-08-04T22:42:23.171653Z","submitted_at":"2026-01-18T18:58:23Z","title":"Agentic Reasoning for Large Language Models","version":1},"reference_index":230,"source":"pdf_text","source_observed_at":"2026-05-17T15:14:25.558878Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2601.12538"},"observation_digest":"sha256:98092187d2735bc69f15e9d1327661ed49b11eb0f40d5e97d37d0335fbb7e495","observation_id":"8565894b-e350-463f-bb4f-9eee26cc4155","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2602.13280","last_updated":"2026-05-05T19:23:23Z","snapshot_observed_at":"2026-08-08T17:22:37.969507Z","submitted_at":"2026-02-06T08:05:15Z","title":"BEAGLE: Behavior-Enforced Agent for Grounded Learner Emulation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T07:09:29.814409Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2602.13280"},"observation_digest":"sha256:6e0b0150ef128d79c787c957c7370f010f2eb761f535547d9d80e2ef2a04cbfb","observation_id":"aadc4c29-1411-45b3-927d-931aa901193b","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-03T01:05:26.197993Z","title":"Understanding the effects of rlhf on llm generalisation and diversity.arXiv preprint arXiv:2310.06452,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.15894","last_updated":"2026-05-27T12:13:06Z","snapshot_observed_at":"2026-08-09T02:18:40.608375Z","submitted_at":"2026-02-11T10:28:50Z","title":"Quality-constrained Entropy Maximization Policy Optimization for LLM Diversity","version":2},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-03T01:05:26.197993Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2602.15894"},"observation_digest":"sha256:b8dd5408f8d3df95b1d2f7e5aab0bfcd3dc72033b8c8a8c9aee8a493135e2f43","observation_id":"b65e7330-a004-4772-9c64-4fb025066304","resolution":{"observed_at":"2026-08-03T01:05:26.197993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-02T21:05:25.374304Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.21492","last_updated":"2026-07-18T04:56:29Z","snapshot_observed_at":"2026-08-11T00:22:38.083129Z","submitted_at":"2026-02-25T01:54:50Z","title":"GradAlign: Gradient-Aligned Data Selection for LLM Reinforcement Learning","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-02T21:05:25.374304Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2602.21492"},"observation_digest":"sha256:57ee5f6d77ac2ab36fda4fe9ec8e00307dee38b0fad7791d2388ccb6798daadd","observation_id":"f3276d2e-e41d-43d1-b9e0-b0ac574d6507","resolution":{"observed_at":"2026-08-02T21:05:25.374304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-07-13T13:07:00.928770Z","title":"Understanding the effects of rlhf on llm generalisation and diversity","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.03557","last_updated":"2026-04-04T02:51:46Z","snapshot_observed_at":"2026-08-09T09:57:29.723666Z","submitted_at":"2026-04-04T02:51:46Z","title":"When Do Hallucinations Arise? A Graph Perspective on the Evolution of Path Reuse and Path Compression","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-13T13:07:00.928770Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2604.03557"},"observation_digest":"sha256:8620866e387cbbab80802b3786f10801b06d6c86099ea98097a1e6eb5bec6505","observation_id":"cfeec28a-5b19-4d3b-b96e-216747cfdd27","resolution":{"observed_at":"2026-07-13T13:07:00.928770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2604.18519","last_updated":"2026-04-20T17:17:07Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:17:07Z","title":"LLM Safety From Within: Detecting Harmful Content with Internal Representations","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-10T04:33:54.058475Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2604.18519"},"observation_digest":"sha256:1abe39a082cf6ff6d8d86692584fe6006195458652f48d0a1038bfb1943bca52","observation_id":"ecedd5a7-16f4-4a84-b195-84538c5d991d","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2604.25634","last_updated":"2026-04-28T13:35:31Z","snapshot_observed_at":"2026-08-11T12:01:25.090022Z","submitted_at":"2026-04-28T13:35:31Z","title":"The Surprising Universality of LLM Outputs: A Real-Time Verification Primitive","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-07T15:44:01.229817Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2604.25634"},"observation_digest":"sha256:4df83a6d8624d4d272a96ff9d7a526972293f4717416711ffc9ba790fb78ce57","observation_id":"b8a0b038-a05c-48f3-9c5b-474805522b71","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.00195","last_updated":"2026-05-11T12:48:16Z","snapshot_observed_at":"2026-07-31T18:51:54.041588Z","submitted_at":"2026-04-30T20:20:59Z","title":"Diversity in Large Language Models under Supervised Fine-Tuning","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-09T20:32:37.788283Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.00195"},"observation_digest":"sha256:f1278588f820847d5f5fcb3190634640d629a28fcfa177161ca594f4e0f5b36f","observation_id":"b263e130-898f-45ae-9a34-1947755b3223","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.00195","last_updated":"2026-05-11T12:48:16Z","snapshot_observed_at":"2026-07-31T18:51:54.041588Z","submitted_at":"2026-04-30T20:20:59Z","title":"Diversity in Large Language Models under Supervised Fine-Tuning","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-12T03:10:22.314719Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.00195"},"observation_digest":"sha256:daa563b3c94404e1656977084e86941882cde2812dc3a298493a323a445d5ece","observation_id":"5ba0c5f4-613e-45ce-ac3d-cfc285c83035","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.01123","last_updated":"2026-05-01T21:49:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-01T21:49:20Z","title":"PERSA: Reinforcement Learning for Professor-Style Personalized Feedback with LLMs","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-09T19:09:07.557773Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.01123"},"observation_digest":"sha256:c18e349cf40967c5d17abb6e714a9de415fd761ee404f46de48410074f6d9fdf","observation_id":"251f3180-c9de-48b4-953b-66ab9af641e1","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.06040","last_updated":"2026-05-07T11:28:53Z","snapshot_observed_at":"2026-08-11T16:00:51.947838Z","submitted_at":"2026-05-07T11:28:53Z","title":"Novelty-based Tree-of-Thought Search for LLM Reasoning and Planning","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-08T10:41:59.816073Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.06040"},"observation_digest":"sha256:5b630ab721bca9f050c1d0ff43f2b8f261f0782afae7bf3c36fb3c5a903ab315","observation_id":"a0c9c9ef-6cfa-4499-961c-058a2d5f6f14","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.06540","last_updated":"2026-05-07T16:38:17Z","snapshot_observed_at":"2026-08-11T21:06:17.406254Z","submitted_at":"2026-05-07T16:38:17Z","title":"Ex Ante Evaluation of AI-Induced Idea Diversity Collapse","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-08T09:49:57.122084Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.06540"},"observation_digest":"sha256:e52c139858b341ab10db4989fbe459b07dbfd6f5506f5492344bbba3d2ae47fe","observation_id":"657e7ec3-a008-475f-81ba-9649c2ec41d8","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.07632","last_updated":"2026-05-08T11:59:34Z","snapshot_observed_at":"2026-07-06T23:20:01.856093Z","submitted_at":"2026-05-08T11:59:34Z","title":"Post-training makes large language models less human-like","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-11T02:47:23.619345Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.07632"},"observation_digest":"sha256:2b715421ccc994b8dc9b1b90471faca60c6bb46b9dc18a1412d1a98c024dec21","observation_id":"95c7f978-6279-4be5-b93a-c4ab1f97c0f7","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.08862","last_updated":"2026-05-09T10:21:38Z","snapshot_observed_at":"2026-08-11T01:48:17.166092Z","submitted_at":"2026-05-09T10:21:38Z","title":"BubbleSpec: Turning Long-Tail Bubbles into Speculative Rollout Drafts for Synchronous Reinforcement Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-12T01:23:13.071107Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.08862"},"observation_digest":"sha256:29e004a3f813cd05a0b9953933627bb3f627a330f7442704b3fcbc0d75f23f81","observation_id":"3f2f4a67-2f64-4c34-8b46-44c4e5e3ebbb","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.09995","last_updated":"2026-05-11T05:11:04Z","snapshot_observed_at":"2026-08-10T20:47:41.204946Z","submitted_at":"2026-05-11T05:11:04Z","title":"Annotations Mitigate Post-Training Mode Collapse","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-12T03:58:11.179607Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.09995"},"observation_digest":"sha256:8361e96a65e9dffd6ba0033f7252521f0feccce32fb0de3340d945952173d642","observation_id":"c87b1bec-0986-4b12-8a16-49e91f333a20","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.10716","last_updated":"2026-05-11T15:25:29Z","snapshot_observed_at":"2026-07-06T23:22:37.355450Z","submitted_at":"2026-05-11T15:25:29Z","title":"What should post-training optimize? A test-time scaling law perspective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T05:20:18.754351Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.10716"},"observation_digest":"sha256:fe2247227ca7d30fb3d98eafd0e25ad171e90f03f6de4c164b5346e36ec5b4d2","observation_id":"9a28c782-0512-4a16-9065-3ac2f1f1208e","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.12522","last_updated":"2026-04-04T17:30:35Z","snapshot_observed_at":"2026-08-11T05:04:35.526400Z","submitted_at":"2026-04-04T17:30:35Z","title":"Differences in Text Generated by Diffusion and Autoregressive Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-14T20:59:06.804446Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.12522"},"observation_digest":"sha256:18605085d859c632a8e69b134cc5a4d44af95ad20be0dbff9d22d6ae6cdf509b","observation_id":"6bc5ddcd-b8b3-408f-a79d-f47128e7124f","resolution":{"observed_at":"2026-05-19T02:34:44.407176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.19976","last_updated":"2026-05-19T15:20:39Z","snapshot_observed_at":"2026-08-07T22:42:18.020690Z","submitted_at":"2026-05-19T15:20:39Z","title":"RECIPE: Procedural Planning via Grounding in Instructional Video","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-20T06:24:50.714859Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.19976"},"observation_digest":"sha256:42a7a8a44e08ea02d0c1be94bda9d7ef58b2716f98798d3d6867503e336b6fa0","observation_id":"d767b407-ae35-441a-bb1c-d4c62e9d66e4","resolution":{"observed_at":"2026-05-20T06:28:05.577866Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2605.28664","last_updated":"2026-05-27T15:59:45Z","snapshot_observed_at":"2026-08-07T19:59:30.749922Z","submitted_at":"2026-05-27T15:59:45Z","title":"Activation Steering for Synthetic Data Generation: The Role of Diversity in Downstream Safety Detection","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T13:53:27.306664Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2605.28664"},"observation_digest":"sha256:5a6723d4a01dc2d929e1f12f5d416b3241f3583492a367a8287a21cb5ff99c03","observation_id":"98e5fd8b-a442-4ba9-b5ff-c5f9e7f8ce5a","resolution":{"observed_at":"2026-06-29T14:03:29.929578Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-07-13T17:56:39.720418Z","title":null,"venue":null,"work_id":null,"year":1985},"citing_paper":{"arxiv_id":"2606.00005","last_updated":"2026-03-26T20:38:21Z","snapshot_observed_at":"2026-08-05T00:30:59.637532Z","submitted_at":"2026-03-26T20:38:21Z","title":"Emergent Collaborative Deliberation in Multi-Model AI Systems: A BFT-Derived Protocol for Epistemic Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-13T17:56:39.720418Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.00005"},"observation_digest":"sha256:d8d6b6b39e98e2cb471517b3a814e0a23e69220ab7caa3c9c8e999c80ba8d884","observation_id":"dc999ee8-daf5-44e8-b90e-f9882bc79575","resolution":{"observed_at":"2026-07-13T17:56:39.720418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2606.01736","last_updated":"2026-06-05T20:16:04Z","snapshot_observed_at":"2026-08-10T02:43:06.686249Z","submitted_at":"2026-06-01T05:58:50Z","title":"Argument Collapse: LLMs Flatten Long-Form Public Debate","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T15:07:55.841128Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.01736"},"observation_digest":"sha256:4439555825857b7a4d8c947a820a7dcf51b493809a166d68daf8555d04618a0f","observation_id":"e1bcf991-daed-4a19-b818-a86891a9a738","resolution":{"observed_at":"2026-07-01T22:46:18.422997Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2606.01811","last_updated":"2026-06-01T07:27:43Z","snapshot_observed_at":"2026-08-01T23:50:35.081089Z","submitted_at":"2026-06-01T07:27:43Z","title":"\"I've Seen How This Goes\": Characterizing Diversity via Progressive Conditional Surprise","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-28T14:51:06.448678Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.01811"},"observation_digest":"sha256:d334cd16cf17c83db54667680c6250ff4d2c04bbe7092c41509872ecdc4d7ff4","observation_id":"ba1f38c7-4133-4276-b6d8-ac37bd6d593e","resolution":{"observed_at":"2026-07-01T22:56:20.811920Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2606.03165","last_updated":"2026-06-02T05:23:45Z","snapshot_observed_at":"2026-07-06T23:43:27.226603Z","submitted_at":"2026-06-02T05:23:45Z","title":"Fully Automated Identification of Lexical Alignment and Preference-Stage Shifts in Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T10:10:05.477340Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.03165"},"observation_digest":"sha256:0e37e2add62c96002cfcb80d4930f7dd34a748992143bd79f0c5a6091eabb488","observation_id":"dfbc2079-4ef1-4f92-a414-02ca0b163678","resolution":{"observed_at":"2026-07-02T03:16:34.829871Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2606.03238","last_updated":"2026-07-09T16:49:37Z","snapshot_observed_at":"2026-08-04T20:21:00.360019Z","submitted_at":"2026-06-02T06:55:52Z","title":"When RLHF Fails: A Mechanistic Taxonomy of Reward Hacking, Collapse, and Evaluator Gaming","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T11:10:25.190293Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.03238"},"observation_digest":"sha256:e0d682023567dcf6cbf6c39017f8811493de2fd477c5bc534be3f7b39e254de8","observation_id":"ffcfde09-32e2-48ca-b1f1-18c34eca2c64","resolution":{"observed_at":"2026-07-02T02:16:26.150703Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-07-12T15:17:34.380336Z","title":"Understanding the effects of RLHF on LLM generalisation and diversity,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.03238","last_updated":"2026-07-09T16:49:37Z","snapshot_observed_at":"2026-08-04T20:21:00.360019Z","submitted_at":"2026-06-02T06:55:52Z","title":"When RLHF Fails: A Mechanistic Taxonomy of Reward Hacking, Collapse, and Evaluator Gaming","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-12T15:17:34.380336Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.03238"},"observation_digest":"sha256:031ab474f94bbb5e390879dfbe4056912c55a3ae4e397988866d6bc1d0027124","observation_id":"705be100-e934-4b3b-8941-0eda957c22d0","resolution":{"observed_at":"2026-07-12T15:17:34.380336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2606.24947","last_updated":"2026-06-23T07:17:58Z","snapshot_observed_at":"2026-08-07T20:33:42.806752Z","submitted_at":"2026-06-23T07:17:58Z","title":"Supervised Reinforcement Learning for the Coordination of Distributed Energy Resources","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T01:05:48.492775Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.24947"},"observation_digest":"sha256:8b2fa698caa0467e9bb85308d870d57eccee72240fb3af733ef0bab8934239af","observation_id":"5d57eb6b-fa60-4771-adba-d0c003d8d1f3","resolution":{"observed_at":"2026-07-04T15:59:57.320099Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":"2310.06452","doi":"10.48550/arxiv.2310.06452","metadata_source":"pith","pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","venue":"cs.LG","work_id":"13d47639-3500-414f-b6ea-1f277577ad3b","year":2023},"citing_paper":{"arxiv_id":"2606.28661","last_updated":"2026-06-27T00:37:33Z","snapshot_observed_at":"2026-08-09T14:28:59.457613Z","submitted_at":"2026-06-27T00:37:33Z","title":"When More Sampling Hurts: The Modal Ceiling and Correlation Ceiling of Test-Time Scaling","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-30T09:44:27.786630Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2606.28661"},"observation_digest":"sha256:5586b8c6d81c68d68a49c76bbf81cc59a0d5f18a34142780779ebbadb985dbb1","observation_id":"496ad41a-b12f-4d54-bebc-57b5843ef3a5","resolution":{"observed_at":"2026-06-30T09:44:37.116492Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-07-12T05:08:55.438431Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03065","last_updated":"2026-07-03T07:57:31Z","snapshot_observed_at":"2026-08-09T04:51:38.797328Z","submitted_at":"2026-07-03T07:57:31Z","title":"Spectral Rewiring for Exploration, Purification, and Model Merging","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-12T05:08:55.438431Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2607.03065"},"observation_digest":"sha256:538caf97709c0250bb832ac164f76911a1e54541c7c8f378f50925d83bfcaeb9","observation_id":"5ead472e-b5ec-42bb-b603-e0297fa82021","resolution":{"observed_at":"2026-07-12T05:08:55.438431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-02T06:23:47.598669Z","title":"Understanding the effects of RLHF on LLM generalisation and diversity","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.12796","last_updated":"2026-07-25T18:18:32Z","snapshot_observed_at":"2026-08-07T16:30:53.506949Z","submitted_at":"2026-07-14T14:12:05Z","title":"The One-Word Census: Answer-Choice Conformity Across 44 Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T06:23:47.598669Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2607.12796"},"observation_digest":"sha256:55a7867efd07f47de4a8fb32a6b0eb4596bbb05c7869fb1eaea0a173d7550e6d","observation_id":"5e2cf7ab-a664-460c-8752-a6e7bfb6a2a4","resolution":{"observed_at":"2026-08-02T06:23:47.598669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-01T15:23:43.844643Z","title":"Understanding the effects of RLHF on LLM generalisation and diversity","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18476","last_updated":"2026-07-20T19:47:16Z","snapshot_observed_at":"2026-08-07T06:13:32.792422Z","submitted_at":"2026-07-20T19:47:16Z","title":"Structured Output Collapses Answer Diversity Across 44 Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T15:23:43.844643Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2607.18476"},"observation_digest":"sha256:49e358793e9adb3c5de998fed37a09947e097d074c98da446247ff654a2de8a0","observation_id":"1338703e-e922-4efa-b958-7c0c22b10bd9","resolution":{"observed_at":"2026-08-01T15:23:43.844643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-01T12:36:11.245333Z","title":"Understanding the effects of rlhf on llm generalisation and diversity.arXiv preprint arXiv:2310.06452,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19523","last_updated":"2026-07-21T19:10:38Z","snapshot_observed_at":"2026-08-06T11:26:29.786981Z","submitted_at":"2026-07-21T19:10:38Z","title":"When Reasoning Narrows the Move: Diversity Collapse in LLM Game Play","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T12:36:11.245333Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2607.19523"},"observation_digest":"sha256:bf926ad8175bf835f5c1e5dce468d0cb03cbd2bda6a5b9c38405a5da7db2b46c","observation_id":"c829a6af-1648-4c32-b739-84b78005ebf6","resolution":{"observed_at":"2026-08-01T12:36:11.245333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-04T00:53:27.653093Z","title":"Understanding the Effects of","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00301","last_updated":"2026-07-31T21:31:07Z","snapshot_observed_at":"2026-08-10T19:46:27.049643Z","submitted_at":"2026-07-31T21:31:07Z","title":"Abstention as an Action Can Kill Both the Reward Gradient and the KL Anchor: Collapse Law and Repair for Error-Penalized Reinforcement Learning","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T00:53:27.653093Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2608.00301"},"observation_digest":"sha256:25ecd80347ac62eaab86298b54c7f6254e75951a6883e9c24fab1ec3f30d42c0","observation_id":"05b2a897-3b9e-419b-aadc-8f88a138c232","resolution":{"observed_at":"2026-08-04T00:53:27.653093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06452","snapshot_observed_at":"2026-08-05T15:25:40.171341Z","title":"Understanding the","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03644","last_updated":"2026-08-04T13:29:00Z","snapshot_observed_at":"2026-08-11T03:36:19.110855Z","submitted_at":"2026-08-04T13:29:00Z","title":"Is Inter-Seed Cross-Play Enough? Evaluating the Robustness of Zero-Shot Coordination Algorithms to Implementation Details","version":1},"reference_index":191,"source":"arxiv_source","source_observed_at":"2026-08-05T15:25:40.171341Z"},"links":{"cited_paper":"/paper/2310.06452","citing_paper":"/paper/2608.03644"},"observation_digest":"sha256:9e4e4a143cce64f29fcf63b37fcb80014ba85dd617c3c523f733f15b59597ecf","observation_id":"a8e36753-a1cb-4b9b-9c2e-3d89feed25e2","resolution":{"observed_at":"2026-08-05T15:25:40.171341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.06452/citation-record","integrity":"/paper/2310.06452/integrity","json":"/paper/2310.06452/citation-record.json","paper":"/paper/2310.06452"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n16-1014","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Diversity-Promoting Objective Function for Neural Conversation Models , booktitle =","venue":null,"work_id":"6ba7c420-c2af-4900-b46e-2289a049768d","year":2016},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:74a3f2459cabc48f1b65bbd68df049ef4f795f4f5bd52fa6b698c303dc17df2f","observation_id":"095f85e7-82d9-412d-ba11-9a7a94f8fd36","resolution":{"observed_at":"2026-05-19T02:34:44.364357Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-18T02:51:35.395533+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T02:51:35.395533+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-08-07T17:14:39.278754Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":"2112.09332","doi":"10.48550/arxiv.2112.09332","metadata_source":"pith","pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebGPT: Browser-assisted question-answering with human feedback","venue":"cs.CL","work_id":"e25ef3e1-4848-4cb9-bf28-67a420591165","year":2021},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:b209f7f3eaf1bc2fbb85be37a06d5872a0e74a42ae769a0f1a125c55ea6c2433","observation_id":"4224ff22-876f-4e9a-a82c-7dbefdd5191b","resolution":{"observed_at":"2026-05-19T02:34:44.369528Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:dfc7561f44e1ce6d9437281ceb8d0c0b74da8d9f40f27009276414acb5b38928","observation_id":"abf215aa-6c84-45b6-9dd2-b083a625335d","resolution":{"observed_at":"2026-05-19T02:34:44.339926Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.01325","last_updated":"2022-02-15T19:09:36Z","snapshot_observed_at":"2026-08-12T00:02:16.697336Z","submitted_at":"2020-09-02T19:54:41Z","title":"Learning to summarize from human feedback","version":3},"cited_work":{"arxiv_id":"2009.01325","doi":"10.18653/v1/2022.naacl-main.6","metadata_source":"pith","pith_arxiv_id":"2009.01325","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning to summarize from human feedback","venue":"cs.CL","work_id":"1fae7759-93bb-4cba-a0d5-ebff515b9d39","year":2020},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2009.01325","citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:80e748d4aae5d1054aeb7096ff324010f0249d34042799a31282116bd67dbdc6","observation_id":"6fb23c8b-53e5-4382-b547-a9523ad1ccb1","resolution":{"observed_at":"2026-05-19T02:34:44.347871Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-07-06T14:33:11.945106Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":"2212.10560","doi":"10.1145/3209978.3210080","metadata_source":"pith","pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","venue":"cs.CL","work_id":"d0018767-775d-406e-861d-539ed681ff73","year":2022},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:00fa2165adf4149ad4778e89f7b5ef8e33567534eae9df686701b0e311cf55b5","observation_id":"c4252e7f-bead-4a96-9bb4-3b91650b414e","resolution":{"observed_at":"2026-05-19T02:34:44.357454Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"believes","venue":null,"work_id":"15ddbbfc-1152-4378-9836-815846086bf7","year":2017},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:73c4103fa82d789d5742ef5d1242a778f731c90c28ddcf5a822886a82c00cc32","observation_id":"ae5a7521-d7f2-47ee-8b76-1cf4381b312c","resolution":{"observed_at":"2026-05-19T02:34:44.395474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f245ba79-2ff5-4fd6-93e1-370f2f7364b0","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:3232619a36ccaeeb3ea1e8b52c208e039be5fb671d4d084ac98aa96c56978f8c","observation_id":"14eae0cf-5233-4aba-bc78-50af528675ba","resolution":{"observed_at":"2026-05-19T02:34:44.398878Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For example, you should combine questions with imperative instrucitons","venue":null,"work_id":"a72acadf-8784-4253-a4ff-42737e564699","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:6a15d0864c646f7393e1abbbb3e6d0c0b6eb3868841eec13677d03f2d103d6c9","observation_id":"ae9bb66e-757d-4790-8f35-758387e370f4","resolution":{"observed_at":"2026-05-19T02:34:44.402357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The list should include diverse types of tasks like open-ended generation, classification, editing, etc","venue":null,"work_id":"c3bf0f65-6741-4832-a50b-890d66cd9e6a","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:73aa168e93a6231d3230936ab13e4c3ec2414d6e422a5878f67e4cfac8644e1f","observation_id":"8253de22-a081-491b-9c13-af0f48f81a8e","resolution":{"observed_at":"2026-05-19T02:34:44.406100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For example, do not ask the assistant to create any visual or audio output","venue":null,"work_id":"4142ec15-1daf-4ad4-97b3-531e2e7ef645","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:57f1fbd43c6bfb6f2b694c6ef8f9194f3628d08b441958330f2f29f950cc3d46","observation_id":"4ad951ff-763c-4ea8-b8aa-efb50821d2fb","resolution":{"observed_at":"2026-05-19T02:34:44.373703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"981694f6-279a-4333-9251-ca866538d42d","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:c06e13db376d50fcb434ed64a434c4e3d415302946de1f5e1d59252c3a3af0f1","observation_id":"13f762b8-98ed-4cf7-9d1f-a8bfdea3c6fd","resolution":{"observed_at":"2026-05-19T02:34:44.377825Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Either an imperative sentence or a question is permitted","venue":null,"work_id":"790d5bb9-ea94-41dd-a8ed-3fde9f895c6d","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:46ea1d769641705265f0500aeee1f0f7b511d3ecd2d294e9009503241ad68e4f","observation_id":"be61dc26-b102-4818-a609-067ee3ae678a","resolution":{"observed_at":"2026-05-19T02:34:44.381454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fbb88b07-fef9-472e-857b-a6d4ca1b4a7f","year":null},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:3200715208a9e6a84c30974b3edf44835c137d00bd2fe0d23a14ccda31077fd1","observation_id":"8c570ad3-3b22-4467-9043-a0b883ad6c82","resolution":{"observed_at":"2026-05-19T02:34:44.384722Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Make sure the output is less than 100 words","venue":null,"work_id":"2869fc34-adc5-4f9d-920a-72b289fffd74","year":2023},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:82b898c2a58f8c522939752a6154f863d25260b31872fd38eacac301508c548a","observation_id":"09dfeb3f-8432-471d-a0ca-3ba804279bf9","resolution":{"observed_at":"2026-05-19T02:34:44.388105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"J.1 D ATASET SPLITTING We create split versions of these datasets along several factors of variation in their inputs: length, sentiment, and subreddit","venue":null,"work_id":"d7bcd4b8-3b88-4dad-add1-bddfc79e9114","year":2013},"citing_paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-19T02:34:44.313274Z"},"links":{"citing_paper":"/paper/2310.06452"},"observation_digest":"sha256:928d5856eeea4717ea203dea9ded44333616002d51e9229d84d15d84ecf3b297","observation_id":"0202a8a2-44c5-4282-b71e-c9fb9e56074b","resolution":{"observed_at":"2026-05-19T02:34:44.392071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2310.06452","last_updated":"2024-02-19T14:39:07Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-29T18:39:02.141391Z","submitted_at":"2023-10-10T09:25:44Z","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":3,"verified_exact":3,"verified_fuzzy":7},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 66 inbound Pith citation observations for arXiv:2310.06452."}