{"as_of":"2026-08-12T13:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6b6f4a0c7566448235c6b219ba7e45fcd9eea4cbabbc3b5f1beb1bc5239d178b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":134,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T04:53:06.389428Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":92,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2304.06767","last_updated":"2023-12-01T14:28:06Z","snapshot_observed_at":"2026-08-07T04:02:54.504761Z","submitted_at":"2023-04-13T18:22:40Z","title":"RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment","version":4},"reference_index":126,"source":"arxiv_source","source_observed_at":"2026-05-18T00:46:56.664582Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2304.06767"},"observation_digest":"sha256:f8970852914122f1f50336b997ec0d8f54df970a0054e07c537396f1e4d03f3c","observation_id":"d8a95642-eca0-4579-8bb7-7b6ef46e5aa6","resolution":{"observed_at":"2026-05-18T00:46:56.825133Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2310.15288","last_updated":"2026-05-07T22:20:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-23T18:54:43Z","title":"Active teacher selection for reward learning","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-24T05:54:43.873174Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2310.15288"},"observation_digest":"sha256:9d799554d68ba9182ae9af032b7e319eb6dbe35f08902dcdc6e5ce6196bbbc7a","observation_id":"a07e6ed0-a364-4ab9-97be-515c3ad34b6b","resolution":{"observed_at":"2026-05-24T05:56:01.733753Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2402.05070","last_updated":"2024-08-20T19:14:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-07T18:21:17Z","title":"A Roadmap to Pluralistic Alignment","version":3},"reference_index":154,"source":"arxiv_source","source_observed_at":"2026-05-16T14:37:53.279275Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2402.05070"},"observation_digest":"sha256:717436168020393335bd6468c1000e44b41c1f2e6bbc879a4b765b8d4d8cc308","observation_id":"828da731-537b-4454-aea0-1201131bcf23","resolution":{"observed_at":"2026-05-16T14:37:53.479800Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-08-10T03:07:35.408836Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"reference_index":140,"source":"arxiv_source","source_observed_at":"2026-05-10T17:34:42.565806Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2403.07974"},"observation_digest":"sha256:39698337a1693453b5eed6a87a087dbefb1097b92b0de262cf4291ff02c3689d","observation_id":"25cf5175-e1be-4c62-b356-a41a5f68d7b9","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2406.10162","last_updated":"2024-06-29T00:28:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-14T16:26:20Z","title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","version":3},"reference_index":264,"source":"arxiv_source","source_observed_at":"2026-05-17T14:43:29.496457Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2406.10162"},"observation_digest":"sha256:288721ec8de8ac8a0d63ac1deb2b7bf50aa7ed1ec58e91b3e9ccbf6f1b7c4942","observation_id":"31597a6d-0474-4ec5-96f5-788952ad2efb","resolution":{"observed_at":"2026-05-17T14:43:30.279272Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2408.12935","last_updated":"2026-05-13T07:56:42Z","snapshot_observed_at":"2026-08-02T12:48:59.218457Z","submitted_at":"2024-08-23T09:33:48Z","title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","version":4},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-05-23T21:54:26.670284Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2408.12935"},"observation_digest":"sha256:467d13ff8de945d71b8fbf738bac5b343efff8bfa3d017897c1369cb0ed0a304","observation_id":"3c04b55c-8786-430e-bb16-57578f39a27c","resolution":{"observed_at":"2026-05-23T21:55:49.986978Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2409.12917","last_updated":"2024-10-04T17:28:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-19T17:16:21Z","title":"Training Language Models to Self-Correct via Reinforcement Learning","version":2},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-05-17T12:04:10.210508Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2409.12917"},"observation_digest":"sha256:8b477ba78478359d0f8f3755c2fb03925c127701bbf615d81c4035fca40a7454","observation_id":"b586926b-b09e-49cc-83be-fd9e878bacb0","resolution":{"observed_at":"2026-05-17T12:04:10.685287Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2410.18451","last_updated":"2024-10-24T06:06:26Z","snapshot_observed_at":"2026-08-07T05:53:12.643515Z","submitted_at":"2024-10-24T06:06:26Z","title":"Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T16:18:01.560780Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2410.18451"},"observation_digest":"sha256:d9c60133d6cd5d4cedae991af7c3610107f0a4a5271aea926842986568b44ba2","observation_id":"d1131658-53a2-4c1b-8da7-02f27aafd9fb","resolution":{"observed_at":"2026-05-17T16:18:01.605404Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-12T04:53:06.389428Z","title":"Michaud, Jacob Pfau, Dmitrii Krasheninnikov, Xin Chen, Lauro Langosco, Peter Hase, Erdem Bıyık, Anca Dragan, David Krueger, Dorsa Sadigh, and Dylan Hadfield-Menell","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.00967","last_updated":"2024-12-01T21:11:28Z","snapshot_observed_at":"2026-08-12T04:46:18.501837Z","submitted_at":"2024-12-01T21:11:28Z","title":"Linear Probe Penalties Reduce LLM Sycophancy","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T04:53:06.389428Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2412.00967"},"observation_digest":"sha256:813e1757932637ce97bdfdd8c2a2ab865f16d302d1791007dd10e799555d0693","observation_id":"b38d39b0-c898-4b59-9a53-8d25da4c4240","resolution":{"observed_at":"2026-08-12T04:53:06.389428Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-11T22:06:54.024699Z","title":"Open Problems and Fundamen- tal Limitations of Reinforcement Learning from Human Feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.03824","last_updated":"2025-06-01T22:36:28Z","snapshot_observed_at":"2026-08-11T21:59:55.905922Z","submitted_at":"2024-12-05T02:37:51Z","title":"Towards Data Governance of Frontier AI Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T22:06:54.024699Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2412.03824"},"observation_digest":"sha256:e08c1871c984c7771720baebed9ac4024d6595d9a8cf14ee79c1f4b4db17608d","observation_id":"00bf9566-314e-43d9-b38e-03e081c3ab0b","resolution":{"observed_at":"2026-08-11T22:06:54.024699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-11T17:19:21.698370Z","title":"K.; Scheurer, J.; Rando, J.; Freedman, R.; Korbak, T.; Lindner, D.; Freire, P.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09173","last_updated":"2024-12-12T11:03:25Z","snapshot_observed_at":"2026-08-12T10:23:56.015788Z","submitted_at":"2024-12-12T11:03:25Z","title":"ReFF: Reinforcing Format Faithfulness in Language Models across Varied Tasks","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T17:19:21.698370Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2412.09173"},"observation_digest":"sha256:05670faa371d1fe0ac4f2cf822f4c76472f2cdb3cd352b5c95f667e3b3d5952d","observation_id":"a718bade-719b-4609-8a77-e6cf523f3bdb","resolution":{"observed_at":"2026-08-11T17:19:21.698370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-11T11:59:02.638813Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.14741","last_updated":"2024-12-19T11:17:31Z","snapshot_observed_at":"2026-08-11T11:54:58.508978Z","submitted_at":"2024-12-19T11:17:31Z","title":"Active Inference and Human--Computer Interaction","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T11:59:02.638813Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2412.14741"},"observation_digest":"sha256:10f2d3a6bdf98444bbd40e929ab453ba091a48af3200c2e672b983298df8d93e","observation_id":"98458dd8-32be-44da-89a8-380a9eab4476","resolution":{"observed_at":"2026-08-11T11:59:02.638813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-11T16:21:35.363100Z","title":"ArXiv, abs/2307.15217","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.15244","last_updated":"2024-12-13T14:18:58Z","snapshot_observed_at":"2026-08-12T02:19:30.084749Z","submitted_at":"2024-12-13T14:18:58Z","title":"MPPO: Multi Pair-wise Preference Optimization for LLMs with Arbitrary Negative Samples","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T16:21:35.363100Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2412.15244"},"observation_digest":"sha256:df4da9a954ce79d550f96e088756f8cdf926951c7e86cebc028e0164d57fb1ed","observation_id":"97938775-c86a-4530-a403-76c05dd628f5","resolution":{"observed_at":"2026-08-11T16:21:35.363100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-11T00:45:35.901723Z","title":"Open problems and fun- damental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.19396","last_updated":"2024-12-27T01:10:17Z","snapshot_observed_at":"2026-08-12T08:33:37.407323Z","submitted_at":"2024-12-27T01:10:17Z","title":"Comparing Few to Rank Many: Active Human Preference Learning using Randomized Frank-Wolfe","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T00:45:35.901723Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2412.19396"},"observation_digest":"sha256:923a654af2d0bdf8f951c1a3a973d2266cb7a85163c23ce7156721a8896a94c9","observation_id":"453f6f5d-cbc5-4d96-a03a-fd10d120fdb0","resolution":{"observed_at":"2026-08-11T00:45:35.901723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T23:06:05.525086Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00083","last_updated":"2024-12-30T16:58:17Z","snapshot_observed_at":"2026-08-11T15:07:21.950361Z","submitted_at":"2024-12-30T16:58:17Z","title":"AI Agent for Education: von Neumann Multi-Agent System Framework","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T23:06:05.525086Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.00083"},"observation_digest":"sha256:6b0b25d51edb4a2cbb2366db013e3c4edbc79880771aa8ce4b7779a329ace8b9","observation_id":"71e97d38-5174-4b94-ba1f-49e4796217cf","resolution":{"observed_at":"2026-08-10T23:06:05.525086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T22:32:56.354957Z","title":"Open problems and fundamental limitations of reinforcemen t learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.01544","last_updated":"2025-01-02T21:31:38Z","snapshot_observed_at":"2026-08-12T10:35:25.091653Z","submitted_at":"2025-01-02T21:31:38Z","title":"Many of Your DPOs are Secretly One: Attempting Unification Through Mutual Information","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T22:32:56.354957Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.01544"},"observation_digest":"sha256:0ad6034dd09a2b2d67f213b749e9a9e184634834fb68be9ed3fce07b224e3b5e","observation_id":"e75027bd-2ea8-4057-830a-9d5dc3d72fa9","resolution":{"observed_at":"2026-08-10T22:32:56.354957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T21:51:07.514828Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.03884","last_updated":"2025-05-30T04:56:02Z","snapshot_observed_at":"2026-08-11T16:24:11.146301Z","submitted_at":"2025-01-07T15:46:42Z","title":"AlphaPO: Reward Shape Matters for LLM Alignment","version":4},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T21:51:07.514828Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.03884"},"observation_digest":"sha256:3f943658eb51d074f3a992ddcfc85d106320562c195991797efe6e39a4c95181","observation_id":"c945e325-1d29-4ab8-b70e-07f7936b83cd","resolution":{"observed_at":"2026-08-10T21:51:07.514828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T21:17:48.341490Z","title":"K.; Scheurer, J.; Rando, J.; Freedman, R.; Korbak, T.; Lindner, D.; Freire, P.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.05336","last_updated":"2025-01-09T16:02:51Z","snapshot_observed_at":"2026-08-11T16:14:42.128959Z","submitted_at":"2025-01-09T16:02:51Z","title":"Stream Aligner: Efficient Sentence-Level Alignment via Distribution Induction","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T21:17:48.341490Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.05336"},"observation_digest":"sha256:1f5ecf49c0e5388a0a3ec13cf06b0387b881da395c5e0e2af0907690887e3bbf","observation_id":"e054cc71-f857-4a29-84a1-efc37b332d2f","resolution":{"observed_at":"2026-08-10T21:17:48.341490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T19:55:50.506817Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.09620","last_updated":"2025-05-29T02:21:03Z","snapshot_observed_at":"2026-08-11T17:33:39.125301Z","submitted_at":"2025-01-16T16:00:37Z","title":"Beyond Reward Hacking: Causal Rewards for Large Language Model Alignment","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T19:55:50.506817Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.09620"},"observation_digest":"sha256:68ec68cf884e0d8bd7a46fcd59f14545779c02a054d0ba6cbe7e9582db4dd6a9","observation_id":"769e38ec-1831-4bfe-a1e9-89e1706b2c4f","resolution":{"observed_at":"2026-08-10T19:55:50.506817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T17:50:56.305623Z","title":"K.; Scheurer, J.; Rando, J.; Freedman, R.; Korbak, T.; Lindner, D.; Freire, P.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13124","last_updated":"2025-01-21T05:36:13Z","snapshot_observed_at":"2026-08-10T22:23:15.426042Z","submitted_at":"2025-01-21T05:36:13Z","title":"Debate Helps Weak-to-Strong Generalization","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T17:50:56.305623Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.13124"},"observation_digest":"sha256:40933d9de6e0e802384b9adf94bb111b7ae6441aaa219b3eb020055825b399e2","observation_id":"a2973a96-1e78-4e54-8240-458dcbcedce9","resolution":{"observed_at":"2026-08-10T17:50:56.305623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-10T13:10:07.857096Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.16461","last_updated":"2025-01-27T19:43:09Z","snapshot_observed_at":"2026-08-11T00:07:00.524750Z","submitted_at":"2025-01-27T19:43:09Z","title":"Trustworthiness in Stochastic Systems: Towards Opening the Black Box","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-10T13:10:07.857096Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.16461"},"observation_digest":"sha256:0b1e4edf431bd766bdbfeb791838da7c5c1c6062cc1cbc1e93c620ec2039cc4e","observation_id":"d8f884b5-31a6-40bb-89b3-0c29f5c8b440","resolution":{"observed_at":"2026-08-10T13:10:07.857096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-09T20:45:51.240862Z","title":"Casper, S., Davies, X., Shi, C., Gilbert, T","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2501.19300","last_updated":"2025-05-29T03:11:10Z","snapshot_observed_at":"2026-08-10T08:15:42.185401Z","submitted_at":"2025-01-31T16:56:18Z","title":"Offline Learning for Combinatorial Multi-armed Bandits","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-09T20:45:51.240862Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.19300"},"observation_digest":"sha256:846c93fbc5880635c55f6bc32532f3e7eb2aa6b05b46d2bc794449ca972070f6","observation_id":"db8e2882-4d5e-4719-b39d-617a0ab8bac5","resolution":{"observed_at":"2026-08-09T20:45:51.240862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-09T20:34:01.238611Z","title":"K., Scheurer, J., Rando, J., Freedman, R., Korbak, T., Lindner, D., Freire, P., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19358","last_updated":"2025-06-02T16:30:23Z","snapshot_observed_at":"2026-08-10T13:03:15.629844Z","submitted_at":"2025-01-31T18:10:53Z","title":"The Energy Loss Phenomenon in RLHF: A New Perspective on Mitigating Reward Hacking","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T20:34:01.238611Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2501.19358"},"observation_digest":"sha256:963aad2cc516ba63cdfd266be19c48b47271ec6f54a417599834b03998219f28","observation_id":"e1bf8590-3090-406b-a708-657b7bd30e25","resolution":{"observed_at":"2026-08-09T20:34:01.238611Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-09T15:10:38.772031Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01715","last_updated":"2025-02-03T16:22:06Z","snapshot_observed_at":"2026-08-12T05:35:24.962597Z","submitted_at":"2025-02-03T16:22:06Z","title":"Process-Supervised Reinforcement Learning for Code Generation","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-09T15:10:38.772031Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.01715"},"observation_digest":"sha256:ba39829927acfdb24b71718b6850ec119afe2936929c40d7e563dff5b34ddb99","observation_id":"81b336dd-1a86-4be9-b198-ad2b53e51433","resolution":{"observed_at":"2026-08-09T15:10:38.772031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-09T01:04:04.115081Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.03717","last_updated":"2025-03-31T23:24:02Z","snapshot_observed_at":"2026-08-11T01:59:53.336838Z","submitted_at":"2025-02-06T02:07:18Z","title":"Efficiently Generating Expressive Quadruped Behaviors via Language-Guided Preference Learning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T01:04:04.115081Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.03717"},"observation_digest":"sha256:8fc16f300454fe1143187cdd0c345ab9f0e44eaf26cb32a4111afc02f7e809d9","observation_id":"19ba90b6-35f8-4d14-996f-9b0f03fd7e62","resolution":{"observed_at":"2026-08-09T01:04:04.115081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-08T22:23:27.507483Z","title":"K., Scheurer, J., Rando, J., Freedman, R., Korbak, T., Lindner, D., Freire, P., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04556","last_updated":"2025-02-06T23:10:14Z","snapshot_observed_at":"2026-08-09T22:38:39.904193Z","submitted_at":"2025-02-06T23:10:14Z","title":"TruthFlow: Truthful LLM Generation via Representation Flow Correction","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T22:23:27.507483Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.04556"},"observation_digest":"sha256:1209d63d2c20bc74693273c49d0505c1c781475d53acd007157661933f6ae906","observation_id":"8f12cd1d-928d-4dad-8eb7-ce13dc188a92","resolution":{"observed_at":"2026-08-08T22:23:27.507483Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-08T20:13:24.242292Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05118","last_updated":"2025-02-07T17:41:29Z","snapshot_observed_at":"2026-08-09T16:18:13.396327Z","submitted_at":"2025-02-07T17:41:29Z","title":"Use of Winsome Robots for Understanding Human Feedback (UWU)","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T20:13:24.242292Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.05118"},"observation_digest":"sha256:f2eb0bd579ddf16eeefb702660a5dba74fb0784d825d5384a2067a5f06bfd714","observation_id":"0a55315f-54a4-494a-955d-74c839955e2f","resolution":{"observed_at":"2026-08-08T20:13:24.242292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-08T20:51:10.056960Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05244","last_updated":"2025-02-07T14:29:07Z","snapshot_observed_at":"2026-08-09T14:58:49.748389Z","submitted_at":"2025-02-07T14:29:07Z","title":"Probabilistic Artificial Intelligence","version":1},"reference_index":2013,"source":"pdf_text","source_observed_at":"2026-08-08T20:51:10.056960Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.05244"},"observation_digest":"sha256:06fb83d21fcab09fe0200711365d4f61d565bbfd9b33ee7fecf947105aefdb4c","observation_id":"ceb3ae33-a214-48ea-a9b7-9747da5c0bc8","resolution":{"observed_at":"2026-08-08T20:51:10.056960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T23:18:39.803268Z","title":"K., Scheurer, J., Rando, J., Freedman, R., Korbak, T., Lindner, D., Freire, P., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08922","last_updated":"2025-02-13T03:15:31Z","snapshot_observed_at":"2026-08-11T00:52:33.304474Z","submitted_at":"2025-02-13T03:15:31Z","title":"Self-Consistency of the Internal Reward Models Improves Self-Rewarding Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T23:18:39.803268Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.08922"},"observation_digest":"sha256:cf3281d72392bb8a7a90d7ee20663ceb8a29f0b16d3a7e59d0e7d7f662ea74cb","observation_id":"1163456b-9200-41ef-9695-08a2fd6e6d56","resolution":{"observed_at":"2026-08-07T23:18:39.803268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T22:21:52.483818Z","title":"K., Scheurer, J., Rando, J., Freedman, R., Korbak, T., Lindner, D., Freire, P., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.09192","last_updated":"2025-05-27T17:24:38Z","snapshot_observed_at":"2026-08-08T23:01:32.478617Z","submitted_at":"2025-02-13T11:32:09Z","title":"Thinking beyond the anthropomorphic paradigm benefits LLM research","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T22:21:52.483818Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2502.09192"},"observation_digest":"sha256:c1d139fdd624c46933b6a1880916f0bd2308a484dd96b1fc9a5685d6d25c1757","observation_id":"7dc7c782-b221-45b8-b989-3fd527b53881","resolution":{"observed_at":"2026-08-07T22:21:52.483818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T15:36:52.805406Z","title":"Michaud, Jacob Pfau, Dmitrii Krasheninnikov, Xin Chen, Lauro Langosco, Peter Hase, Erdem Bıyık, Anca Dragan, David Krueger, Dorsa Sadigh, and Dylan Hadfield-Menell","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14758","last_updated":"2025-05-20T16:28:00Z","snapshot_observed_at":"2026-08-09T15:38:02.951301Z","submitted_at":"2025-05-20T16:28:00Z","title":"Kaleidoscope Gallery: Exploring Ethics and Generative AI Through Art","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:36:52.805406Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2505.14758"},"observation_digest":"sha256:bc49d7d227d6a863b8b86bff96a6ab8a60985ce8bb998cc99dc13badbb277185","observation_id":"6c96f6ae-cab5-45d5-bba6-5fbf314e9818","resolution":{"observed_at":"2026-08-07T15:36:52.805406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T15:19:08.255598Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15694","last_updated":"2025-05-21T16:07:47Z","snapshot_observed_at":"2026-08-09T15:55:39.140815Z","submitted_at":"2025-05-21T16:07:47Z","title":"A Unified Theoretical Analysis of Private and Robust Offline Alignment: from RLHF to DPO","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T15:19:08.255598Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2505.15694"},"observation_digest":"sha256:ff01bd260f9f214931de3a40990bedb35510cbb44f82777d29cd6fb010eac1e3","observation_id":"9bf3d758-8c50-40ea-88e2-3edf4d488e02","resolution":{"observed_at":"2026-08-07T15:19:08.255598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T14:26:34.921482Z","title":"Michaud, Jacob Pfau, Dmitrii Krasheninnikov, Xin Chen, Lauro Langosco, Peter Hase, Erdem Bıyık, Anca Dragan, David Krueger, Dorsa Sadigh, and Dylan Hadfield-Menell","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18893","last_updated":"2025-05-30T14:09:51Z","snapshot_observed_at":"2026-08-09T08:24:03.844078Z","submitted_at":"2025-05-24T22:35:32Z","title":"Reality Check: A New Evaluation Ecosystem Is Necessary to Understand AI's Real World Effects","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:26:34.921482Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2505.18893"},"observation_digest":"sha256:9f52d7c548df40b987cf57f145dedd2960b6f7db058a4e1fbfc44cf9effde978","observation_id":"c234d425-3000-4e85-8a67-7c03b38df923","resolution":{"observed_at":"2026-08-07T14:26:34.921482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T13:44:55.826681Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21395","last_updated":"2025-05-27T16:23:24Z","snapshot_observed_at":"2026-08-07T13:26:21.678395Z","submitted_at":"2025-05-27T16:23:24Z","title":"Square$\\chi$PO: Differentially Private and Robust $\\chi^2$-Preference Optimization in Offline Direct Alignment","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T13:44:55.826681Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2505.21395"},"observation_digest":"sha256:56e50223f32f4888415612313681038079eeacd6ea02438be9a78cc9c2e4c79d","observation_id":"a629e6ad-7be2-4c03-81fd-d1473e5d3b6d","resolution":{"observed_at":"2026-08-07T13:44:55.826681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T13:24:31.042608Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21907","last_updated":"2025-05-31T04:48:02Z","snapshot_observed_at":"2026-08-09T21:30:35.237559Z","submitted_at":"2025-05-28T02:52:39Z","title":"Modeling and Optimizing User Preferences in AI Copilots: A Comprehensive Survey and Taxonomy","version":2},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-08-07T13:24:31.042608Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2505.21907"},"observation_digest":"sha256:19dfbf46ab9015814b508ff6434f65de387ef1aff2c9fb010ae8ba2a47a2cd2d","observation_id":"b351382a-f6ee-491e-9b37-d8518700b8c6","resolution":{"observed_at":"2026-08-07T13:24:31.042608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T12:43:47.614974Z","title":"Michaud, Jacob Pfau, Dmitrii Krasheninnikov, Xin Chen, Lauro Langosco, Peter Hase, Erdem Bıyık, Anca Dragan, David Krueger, Dorsa Sadigh, and Dylan Hadfield-Menell","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23927","last_updated":"2025-05-29T18:22:02Z","snapshot_observed_at":"2026-08-09T08:37:01.244014Z","submitted_at":"2025-05-29T18:22:02Z","title":"Thompson Sampling in Online RLHF with General Function Approximation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:43:47.614974Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2505.23927"},"observation_digest":"sha256:01b4a7b7b22d4a9fe7046517ad3ddec89f98034b478e5718c31b038a21dd62a9","observation_id":"1c0b73b2-aa5c-49b1-990e-9906ac6bb7ed","resolution":{"observed_at":"2026-08-07T12:43:47.614974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T13:10:11.081752Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00047","last_updated":"2025-05-28T16:52:44Z","snapshot_observed_at":"2026-08-07T13:01:23.937258Z","submitted_at":"2025-05-28T16:52:44Z","title":"Risks of AI-driven product development and strategies for their mitigation","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T13:10:11.081752Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.00047"},"observation_digest":"sha256:acf8da00d602c8547b386a42029e2f75cb407bb21bcb64e5d7d3db766b335c87","observation_id":"aaebfc88-f326-44ee-abbd-07aab08942f9","resolution":{"observed_at":"2026-08-07T13:10:11.081752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T10:57:42.237392Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03827","last_updated":"2025-06-04T10:57:18Z","snapshot_observed_at":"2026-08-08T01:31:46.984934Z","submitted_at":"2025-06-04T10:57:18Z","title":"Multi-objective Aligned Bidword Generation Model for E-commerce Search Advertising","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:57:42.237392Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.03827"},"observation_digest":"sha256:38e5b32013ed1ba0bcb909eca84e2b0894e2773dd8bb075b18ce607aec8190f2","observation_id":"1b0746f2-8130-4e13-81be-37aa6ee289b5","resolution":{"observed_at":"2026-08-07T10:57:42.237392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T10:54:09.014935Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04063","last_updated":"2025-06-04T15:26:38Z","snapshot_observed_at":"2026-08-11T19:09:43.027428Z","submitted_at":"2025-06-04T15:26:38Z","title":"Crowd-SFT: Crowdsourcing for LLM Alignment","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:54:09.014935Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.04063"},"observation_digest":"sha256:24f8b79d8a967edae1d870001b7994cb5d6234942306427eb0cb37140b19bdb2","observation_id":"ef7b58ca-d551-4fcd-bb97-aa7a6aee3c69","resolution":{"observed_at":"2026-08-07T10:54:09.014935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T05:48:18.687477Z","title":"K., Scheurer, J., Rando, J.,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11110","last_updated":"2025-06-08T14:08:22Z","snapshot_observed_at":"2026-08-07T05:38:21.402950Z","submitted_at":"2025-06-08T14:08:22Z","title":"AssertBench: A Benchmark for Evaluating Self-Assertion in Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:48:18.687477Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.11110"},"observation_digest":"sha256:ca870695d0e8602a1134a175b74d61c6611d5fe84d7a572024a1c5939b793691","observation_id":"22e083f8-5051-4f3b-aa9e-db05d3809bee","resolution":{"observed_at":"2026-08-07T05:48:18.687477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2506.12382","last_updated":"2026-04-27T08:32:16Z","snapshot_observed_at":"2026-08-02T04:53:57.144183Z","submitted_at":"2025-06-14T07:31:52Z","title":"Exploring the Secondary Risks of Large Language Models","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T09:40:58.067398Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.12382"},"observation_digest":"sha256:7e347a7c832ec4541f9b38591ba5be9e24a342fa1397906b529943ba1fbd64f8","observation_id":"a262c9d3-2641-4e1a-af74-b2c73504f47d","resolution":{"observed_at":"2026-05-19T09:42:14.023050Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T05:43:07.123823Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.13774","last_updated":"2025-08-08T20:29:52Z","snapshot_observed_at":"2026-08-08T03:32:46.630570Z","submitted_at":"2025-06-08T20:31:26Z","title":"Personalized Constitutionally-Aligned Agentic Superego: Secure AI Behavior Aligned to Diverse Human Values","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:43:07.123823Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.13774"},"observation_digest":"sha256:950d97bac0dadd549e557d551d14502657cd3189fbe16a5265314d0e163e85f0","observation_id":"cab35d18-a447-4352-ac03-fd37e5153ac7","resolution":{"observed_at":"2026-08-07T05:43:07.123823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T00:24:12.912770Z","title":"K., Scheurer, J., Rando, J., Freedman, R., Korbak, T., Lindner, D., Freire, P., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.14146","last_updated":"2025-06-17T03:20:41Z","snapshot_observed_at":"2026-08-11T10:40:06.466250Z","submitted_at":"2025-06-17T03:20:41Z","title":"Collaborative Editable Model","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:12.912770Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.14146"},"observation_digest":"sha256:db2626d43f8e55fdefab0e668230d1dd30d9572b6dd027a1c33ea00d5e7bf76c","observation_id":"cddfd517-9ac4-4910-a2d7-25882bebf5e6","resolution":{"observed_at":"2026-08-07T00:24:12.912770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-07T00:24:28.071431Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.14166","last_updated":"2025-06-17T04:06:45Z","snapshot_observed_at":"2026-08-11T12:27:46.731678Z","submitted_at":"2025-06-17T04:06:45Z","title":"Affective-CARA: A Knowledge Graph Driven Framework for Culturally Adaptive Emotional Intelligence in HCI","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T00:24:28.071431Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.14166"},"observation_digest":"sha256:acc89e281be9e78acf4839675a23729b860750541cbe9960488baf4148117506","observation_id":"df5fe3e3-889a-428e-a270-c6de61da79d2","resolution":{"observed_at":"2026-08-07T00:24:28.071431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2506.21834","last_updated":"2026-04-12T01:02:59Z","snapshot_observed_at":"2026-08-02T05:26:21.940391Z","submitted_at":"2025-06-27T00:47:07Z","title":"PrefPaint: Enhancing Medical Image Inpainting through Expert Human Feedback","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-19T08:31:24.549747Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2506.21834"},"observation_digest":"sha256:508a0c533e8c4c3825e4ffa2318e47d3a513c6f445ad97f8a78a24ad7ec7e580","observation_id":"1dd5e96f-9931-4ee4-84e8-729017ed3d0f","resolution":{"observed_at":"2026-05-19T08:32:11.600714Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T20:18:15.468808Z","title":"Casper et al., ‘Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback’, Sep","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.03409","last_updated":"2025-07-04T09:16:11Z","snapshot_observed_at":"2026-08-08T10:46:44.107781Z","submitted_at":"2025-07-04T09:16:11Z","title":"Lessons from a Chimp: AI \"Scheming\" and the Quest for Ape Language","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:18:15.468808Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.03409"},"observation_digest":"sha256:c7be919f7da8a65a109a2f6ed5244b98d867b57ef593c9eb45f888943bb87966","observation_id":"77d01e76-e017-48ab-9720-a3eca423d311","resolution":{"observed_at":"2026-08-06T20:18:15.468808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2507.04005","last_updated":"2026-04-05T08:35:28Z","snapshot_observed_at":"2026-08-02T12:46:52.977974Z","submitted_at":"2025-07-05T11:17:20Z","title":"Exploring a Gamified Personality Assessment Method through Interaction with LLM Agents Embodying Different Personalities","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-19T06:35:06.890058Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.04005"},"observation_digest":"sha256:9c68e58f0ddda495467b14f87206c470495319c8d7c507ce0e20ec5bbbd04a0d","observation_id":"1cd622ca-9fb5-4544-a414-a41cc0fa6ecd","resolution":{"observed_at":"2026-05-19T06:37:07.656554Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T19:28:53.661417Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05913","last_updated":"2025-07-08T11:59:48Z","snapshot_observed_at":"2026-08-09T02:51:17.508221Z","submitted_at":"2025-07-08T11:59:48Z","title":"Best-of-N through the Smoothing Lens: KL Divergence and Regret Analysis","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T19:28:53.661417Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.05913"},"observation_digest":"sha256:38ca7aaa21d9af4f3b00b356e0314272bd8c002a0d9bf54c5769290690fb4027","observation_id":"ac2747d2-67f7-42a0-9154-a4e87dbab96e","resolution":{"observed_at":"2026-08-06T19:28:53.661417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T17:01:18.225322Z","title":"K., Scheurer, J., Rando, J., Freedman, R., Korbak, T., Lindner, D., Freire, P., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.12041","last_updated":"2025-07-16T08:58:27Z","snapshot_observed_at":"2026-08-12T10:57:37.771699Z","submitted_at":"2025-07-16T08:58:27Z","title":"Granular feedback merits sophisticated aggregation","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T17:01:18.225322Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.12041"},"observation_digest":"sha256:032ab2942549de3cd820155baf31e38d807fb1b8c3941cbe66071902e7ef2b9d","observation_id":"79a2f52d-fd27-4fbb-9f10-74f12a55cf3c","resolution":{"observed_at":"2026-08-06T17:01:18.225322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T16:27:57.204698Z","title":"Michaud, Jacob Pfau, Dmitrii Krasheninnikov, Xin Chen, Lauro Langosco, Peter Hase, Erdem Bıyık, Anca Dragan, David Krueger, Dorsa Sadigh, and Dylan Hadfield-Menell","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13541","last_updated":"2025-07-17T21:21:54Z","snapshot_observed_at":"2026-08-12T02:10:46.815006Z","submitted_at":"2025-07-17T21:21:54Z","title":"PrefPalette: Personalized Preference Modeling with Latent Attributes","version":1},"reference_index":1952,"source":"pdf_text","source_observed_at":"2026-08-06T16:27:57.204698Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.13541"},"observation_digest":"sha256:aad2f3c4d2eccd5859d9ce84e27d98a255607d4256432d15f9b7073e9879cc0f","observation_id":"81e681ba-1912-40da-ba7b-1d2f75ac1ce6","resolution":{"observed_at":"2026-08-06T16:27:57.204698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T17:35:48.002781Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14202","last_updated":"2025-07-14T17:41:12Z","snapshot_observed_at":"2026-08-08T03:22:49.699348Z","submitted_at":"2025-07-14T17:41:12Z","title":"PRM-Free Security Alignment of Large Models via Red Teaming and Adversarial Training","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T17:35:48.002781Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.14202"},"observation_digest":"sha256:d95386986a1c480f96c92bed50599319149f95d571b71617b94365b1bb898687","observation_id":"8a5a0256-0ed2-4a11-a324-0a6ae5dfe732","resolution":{"observed_at":"2026-08-06T17:35:48.002781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T15:37:50.967677Z","title":"Casper, X","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15903","last_updated":"2026-07-03T05:12:16Z","snapshot_observed_at":"2026-08-11T23:50:04.262701Z","submitted_at":"2025-07-21T09:08:58Z","title":"Towards Mitigation of Hallucination for LLM-empowered Agents: Progressive Generalization Bound Exploration and Watchdog Monitor","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T15:37:50.967677Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.15903"},"observation_digest":"sha256:6ae1c60f4fb74ab473d216ed34cb90e6a86111fd8526819860c1cadffdd936da","observation_id":"5b05867a-b2bb-41e9-bdae-b1cfd9d5b789","resolution":{"observed_at":"2026-08-06T15:37:50.967677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-06T11:16:04.349891Z","title":"Casper, X","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.22879","last_updated":"2025-07-31T16:54:43Z","snapshot_observed_at":"2026-08-06T11:16:02.241906Z","submitted_at":"2025-07-30T17:55:06Z","title":"RecGPT Technical Report","version":2},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-06T11:16:04.349891Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2507.22879"},"observation_digest":"sha256:452f93049cf19e824d5d73f6a76cc4e4d41079846425f0603ab4c39fd02414bd","observation_id":"e8b2e8f2-8d63-476c-91d6-e9393a28425a","resolution":{"observed_at":"2026-08-06T11:16:04.349891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2508.04149","last_updated":"2026-05-16T09:55:19Z","snapshot_observed_at":"2026-07-06T22:08:36.543090Z","submitted_at":"2025-08-06T07:24:14Z","title":"Difficulty-Based Preference Data Selection by DPO Implicit Reward Gap","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-21T23:46:24.208438Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2508.04149"},"observation_digest":"sha256:a46c042ac6989d05543eb84ebecd7410ddf7687c78735a6fc463cf19f966d84a","observation_id":"5d235c00-dbf4-4475-9b75-70eb77925cd5","resolution":{"observed_at":"2026-05-21T23:50:47.581965Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T15:08:47.910667Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21101","last_updated":"2025-08-28T07:05:24Z","snapshot_observed_at":"2026-08-07T09:02:43.747811Z","submitted_at":"2025-08-28T07:05:24Z","title":"Beyond Prediction: Reinforcement Learning as the Defining Leap in Healthcare AI","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T15:08:47.910667Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2508.21101"},"observation_digest":"sha256:6217aa93f6f808a09fd486c32967f701c901290409aa258cf90e55caf3d024b5","observation_id":"05948fb1-cc9e-4c62-833a-44cbbf013ac9","resolution":{"observed_at":"2026-08-05T15:08:47.910667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T12:53:38.242707Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback.arXiv preprint arXiv:2307.15217, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01181","last_updated":"2025-09-01T07:06:36Z","snapshot_observed_at":"2026-08-06T10:09:29.462010Z","submitted_at":"2025-09-01T07:06:36Z","title":"FocusDPO: Dynamic Preference Optimization for Multi-Subject Personalized Image Generation via Adaptive Focus","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T12:53:38.242707Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2509.01181"},"observation_digest":"sha256:56af0c55d3134fd8ba9d8ad9fcfb382861fb9135b2597594719b57b2df2c003d","observation_id":"dd4f5822-fdd6-4f1e-b279-7c9a34e7a1ff","resolution":{"observed_at":"2026-08-05T12:53:38.242707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T10:54:10.974544Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03672","last_updated":"2025-09-03T19:42:50Z","snapshot_observed_at":"2026-08-08T09:32:45.744670Z","submitted_at":"2025-09-03T19:42:50Z","title":"SharedRep-RLHF: A Shared Representation Approach to RLHF with Diverse Preferences","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T10:54:10.974544Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2509.03672"},"observation_digest":"sha256:c17aa655bb3b8770ab83ca690363d031008440015dcd1324d6fdd83ee028b0a0","observation_id":"97bb6238-1d2c-4042-b3e8-2ea3e4dddebe","resolution":{"observed_at":"2026-08-05T10:54:10.974544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T12:04:04.797646Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10509","last_updated":"2025-09-02T05:46:28Z","snapshot_observed_at":"2026-08-10T23:45:49.170876Z","submitted_at":"2025-09-02T05:46:28Z","title":"The Anti-Ouroboros Effect: Emergent Resilience in Large Language Models from Recursive Selective Feedback","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T12:04:04.797646Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2509.10509"},"observation_digest":"sha256:644830e300a5c99a7a8ade35f56713f59c3096e7286acb45dd2b19fdc7b0fd9e","observation_id":"204f11e3-f7e5-475c-9fe1-17de2c24ec84","resolution":{"observed_at":"2026-08-05T12:04:04.797646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-04T19:19:22.387405Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.10570","last_updated":"2025-09-11T10:30:06Z","snapshot_observed_at":"2026-08-09T17:39:56.182010Z","submitted_at":"2025-09-11T10:30:06Z","title":"Large Foundation Models for Trajectory Prediction in Autonomous Driving: A Comprehensive Survey","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T19:19:22.387405Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2509.10570"},"observation_digest":"sha256:40c882701d1980d55b3d65e959bac3136d7aca5a74896fa7ac361518be1b18eb","observation_id":"95956f12-4d1f-464e-a2b7-b46706ebc560","resolution":{"observed_at":"2026-08-04T19:19:22.387405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-04T14:44:13.604742Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.23730","last_updated":"2026-05-28T14:53:43Z","snapshot_observed_at":"2026-08-09T17:44:40.916466Z","submitted_at":"2025-09-28T08:20:22Z","title":"EAPO: Enhancing Policy Optimization with On-Demand Expert Assistance","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T14:44:13.604742Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2509.23730"},"observation_digest":"sha256:e8d59f358965f73d367bfb77fb99ecad67f24920f53db7875b1babd2badb6c82","observation_id":"a2f5499a-556a-4cb9-b502-68a2b81a64ec","resolution":{"observed_at":"2026-08-04T14:44:13.604742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-04T11:17:36.700516Z","title":"Open problems and fundamental limitations of reinforcement learning from human feedback.arXiv preprint arXiv:2307.15217,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.06096","last_updated":"2026-06-26T18:48:41Z","snapshot_observed_at":"2026-08-09T20:20:11.286361Z","submitted_at":"2025-10-07T16:25:14Z","title":"The Alignment Auditor: A Bayesian Framework for Verifying and Refining LLM Objectives","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T11:17:36.700516Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2510.06096"},"observation_digest":"sha256:db57b356a6c29f051566404a96f76b16613032330cb8c215952b66e0b164b3d1","observation_id":"b2174ce8-2c66-4898-a84a-1f6dcd1579e8","resolution":{"observed_at":"2026-08-04T11:17:36.700516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-03T16:19:31.037475Z","title":"arXiv preprint arXiv:2307.15217 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2512.15792","last_updated":"2026-06-16T04:26:41Z","snapshot_observed_at":"2026-08-12T01:09:53.296319Z","submitted_at":"2025-12-16T03:38:08Z","title":"A Multifaceted Analysis of Social Biases in Large Language Models","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-03T16:19:31.037475Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2512.15792"},"observation_digest":"sha256:2704fce9473169bec8a03d9f4793b0a2f25117a5aa908eaa0c39779a0838ae4b","observation_id":"4bec7b9b-952a-4327-a2fe-21ec7b80fcf2","resolution":{"observed_at":"2026-08-03T16:19:31.037475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2512.21110","last_updated":"2026-04-24T20:27:34Z","snapshot_observed_at":"2026-07-06T22:39:58.137482Z","submitted_at":"2025-12-24T11:15:57Z","title":"Beyond Context: Large Language Models' Failure to Grasp Users' Intent","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T20:09:25.827452Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2512.21110"},"observation_digest":"sha256:851660eb64cb8d4bbebafa623397aef5f407d655be1bf4441214804abb2e7303","observation_id":"1e93959a-afe3-4f8a-977d-a8e32867b405","resolution":{"observed_at":"2026-05-16T20:11:13.642122Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2601.21484","last_updated":"2026-05-19T09:15:43Z","snapshot_observed_at":"2026-08-11T06:35:24.271244Z","submitted_at":"2026-01-29T10:06:52Z","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T09:37:57.120779Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2601.21484"},"observation_digest":"sha256:2204e6140804d699f5ebd22cd5e3fb56c2ace3dfed9d1caa3ea0901b2fe82925","observation_id":"ee760c3f-a2e5-4896-b32f-a326426bfb41","resolution":{"observed_at":"2026-05-16T09:40:49.108927Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2601.21484","last_updated":"2026-05-19T09:15:43Z","snapshot_observed_at":"2026-08-11T06:35:24.271244Z","submitted_at":"2026-01-29T10:06:52Z","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T14:05:37.120262Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2601.21484"},"observation_digest":"sha256:a60cdd599af1c610c15cfd38e1537aaa3aea36a3f3364b1f02b6cd4d27f3a300","observation_id":"114285a7-d4c2-4d75-b1a5-181c265bfa08","resolution":{"observed_at":"2026-05-21T14:10:13.449970Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-03T06:49:03.255485Z","title":"Casper et al., Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback, arXiv:2307.15217","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.21881","last_updated":"2026-06-06T16:30:21Z","snapshot_observed_at":"2026-08-09T11:32:18.790034Z","submitted_at":"2026-01-29T15:46:29Z","title":"Acquiring Human-Like Data-Efficient Mechanics Prediction from Deep Reinforcement Learning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T06:49:03.255485Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2601.21881"},"observation_digest":"sha256:efcf7eadbc817dc52898a51a1273fc2b58f58d2b92c7200144be387bbafee8b3","observation_id":"0e13fa8f-ac8b-46b0-a53d-0f724e22b70f","resolution":{"observed_at":"2026-08-03T06:49:03.255485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-03T04:54:12.936609Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.04000","last_updated":"2026-08-04T16:43:12Z","snapshot_observed_at":"2026-08-07T23:11:25.596094Z","submitted_at":"2026-02-03T20:37:59Z","title":"After Talking with 1,000 Personas: Learning Preference-Aligned Proactive Assistants From Large-Scale Persona Interactions","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T04:54:12.936609Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2602.04000"},"observation_digest":"sha256:a00dd4a7376f928d707810d50830b536b390b5c2e89b94acd38a7d1e155da9d9","observation_id":"14962ea5-6d68-42a0-b81d-f9a56741eaab","resolution":{"observed_at":"2026-08-03T04:54:12.936609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2602.19837","last_updated":"2026-05-06T07:57:45Z","snapshot_observed_at":"2026-08-11T09:31:30.100145Z","submitted_at":"2026-02-23T13:39:58Z","title":"Meta-Learning and Meta-Reinforcement Learning -- Tracing the Path towards DeepMind's Adaptive Agent","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T20:46:15.275441Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2602.19837"},"observation_digest":"sha256:bc6c594ac78fdb5a424c848ac70ec51cabeb345c23f128078c9229d56625179d","observation_id":"2c3f6a3f-121e-4ebc-bd54-8368f0cd73ba","resolution":{"observed_at":"2026-05-15T20:46:35.697989Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.02686","last_updated":"2026-04-03T03:30:34Z","snapshot_observed_at":"2026-08-02T18:52:24.780545Z","submitted_at":"2026-04-03T03:30:34Z","title":"Beyond Semantic Manipulation: Token-Space Attacks on Reward Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T20:29:31.354743Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.02686"},"observation_digest":"sha256:e0017284211b84a8fac58b9b316505b700dc0093416eca736be329be293b9f3d","observation_id":"42c1d9f3-25d4-4821-a6a5-414e6fc24da4","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.06621","last_updated":"2026-04-08T03:01:58Z","snapshot_observed_at":"2026-07-06T22:55:01.863342Z","submitted_at":"2026-04-08T03:01:58Z","title":"The Theorems of Dr. David Blackwell and Their Contributions to Artificial Intelligence","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T18:33:50.296933Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.06621"},"observation_digest":"sha256:330ba3e4f002655e544ab38132db0af2db304f52e6c08bb6314394a9f31659c0","observation_id":"8b647c86-fe95-4a56-842f-3bf8c19fe435","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.07754","last_updated":"2026-04-09T03:20:29Z","snapshot_observed_at":"2026-07-06T22:57:00.904627Z","submitted_at":"2026-04-09T03:20:29Z","title":"The Art of (Mis)alignment: How Fine-Tuning Methods Effectively Misalign and Realign LLMs in Post-Training","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:18:56.476698Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.07754"},"observation_digest":"sha256:3950dc8fc0f2e05d484487fc81c32482eb4930402e33e1a6e6e54de66dfd752d","observation_id":"2e66137a-7514-48f5-b823-6564a678bd7a","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.10134","last_updated":"2026-04-11T09:59:46Z","snapshot_observed_at":"2026-08-11T15:42:26.029363Z","submitted_at":"2026-04-11T09:59:46Z","title":"PlanGuard: Defending Agents against Indirect Prompt Injection via Planning-based Consistency Verification","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T16:19:09.341399Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.10134"},"observation_digest":"sha256:a84d847392760c9027dc523763e1f9854227c5fcb63110e8ed293fae703f5fb8","observation_id":"f2b2cbde-4eca-4447-9353-aeacf297aa06","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.13803","last_updated":"2026-04-15T12:38:51Z","snapshot_observed_at":"2026-08-11T11:39:16.627160Z","submitted_at":"2026-04-15T12:38:51Z","title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-10T13:09:35.407790Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.13803"},"observation_digest":"sha256:de02d5e3a09e36c879debd4c25aca7666d3b5b7264a4b719ee2c18910e80d49e","observation_id":"13c48fb7-dd71-4edc-8adc-4923ed9218ac","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.17587","last_updated":"2026-04-19T19:32:52Z","snapshot_observed_at":"2026-08-02T05:24:23.197254Z","submitted_at":"2026-04-19T19:32:52Z","title":"AIRA: AI-Induced Risk Audit: A Structured Inspection Framework for AI-Generated Code","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-10T05:14:19.185184Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.17587"},"observation_digest":"sha256:f905b14039feffe94bd5f5aa485a6383d498c6c6f91be72f882b0111b6b9db4a","observation_id":"36751cd0-2cc4-4a3e-8169-3fa3be83a811","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.19024","last_updated":"2026-04-21T03:20:07Z","snapshot_observed_at":"2026-08-10T22:10:34.698901Z","submitted_at":"2026-04-21T03:20:07Z","title":"Policy Gradient Primal-Dual Method for Safe Reinforcement Learning from Human Feedback","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-10T02:23:20.208976Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.19024"},"observation_digest":"sha256:277a6172bbca59c895113e15e7044b59eace9ea67bf5ef677d7fe1b41d57a00e","observation_id":"f7d9e837-e024-45cd-b794-36bf595eac1b","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.21216","last_updated":"2026-04-23T02:13:29Z","snapshot_observed_at":"2026-08-05T15:05:21.369016Z","submitted_at":"2026-04-23T02:13:29Z","title":"Post-AGI Economies: Autonomy and the First Fundamental Theorem of Welfare Economics","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-08T13:13:01.920636Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.21216"},"observation_digest":"sha256:75fcb9162eaf27cfa2bd3cd77cf3dba66b7354dbae159ca18670d924adad5dfa","observation_id":"21963a03-975b-4c12-8210-0430445ba567","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.25895","last_updated":"2026-04-28T17:39:14Z","snapshot_observed_at":"2026-07-06T23:11:39.318906Z","submitted_at":"2026-04-28T17:39:14Z","title":"Three Models of RLHF Annotation: Extension, Evidence, and Authority","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-07T14:28:31.451466Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.25895"},"observation_digest":"sha256:96f7b08eae0d09b564c0f86851fdfed6d791b2ca6453552a5ee4aeaa40b1c18c","observation_id":"d85e55c0-fb36-4e7e-8db8-167d471bb613","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.28010","last_updated":"2026-05-15T20:10:07Z","snapshot_observed_at":"2026-07-06T23:13:20.516759Z","submitted_at":"2026-04-30T15:30:47Z","title":"Learning from Disagreement: Clinician Overrides as Implicit Preference Signals for Clinical AI in Value-Based Care","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-07T06:26:57.714846Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.28010"},"observation_digest":"sha256:fad568cf21eb606d881d7ad1f74507b007f11c97ff356986d42066c20c75b784","observation_id":"f4fbdf6f-ea2f-4df3-9781-2c01b6941981","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2604.28010","last_updated":"2026-05-15T20:10:07Z","snapshot_observed_at":"2026-07-06T23:13:20.516759Z","submitted_at":"2026-04-30T15:30:47Z","title":"Learning from Disagreement: Clinician Overrides as Implicit Preference Signals for Clinical AI in Value-Based Care","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-20T23:43:37.688604Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2604.28010"},"observation_digest":"sha256:7ba0dd531e6011cb05f82b40495f655b2e9bd98207f74847f8f1b3c61187ff87","observation_id":"594bfc7a-b5ee-4499-9129-3b49da21f9ec","resolution":{"observed_at":"2026-05-20T23:43:50.990306Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.01954","last_updated":"2026-05-03T16:37:52Z","snapshot_observed_at":"2026-08-11T17:52:37.719010Z","submitted_at":"2026-05-03T16:37:52Z","title":"Moira: Language-driven Hierarchical Reinforcement Learning for Pair Trading","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-09T17:08:46.405278Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.01954"},"observation_digest":"sha256:2a84048dcdade1d887f1bc11ac7bccc38950f1f1d7b3583e9bde33312aa6c675","observation_id":"92b09ebb-7d35-401e-935b-ef0397d63fbc","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.02495","last_updated":"2026-05-25T07:21:41Z","snapshot_observed_at":"2026-08-11T12:31:27.931004Z","submitted_at":"2026-05-04T11:45:38Z","title":"Efficient Preference Poisoning Attack on Offline RLHF","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-08T19:29:25.000361Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.02495"},"observation_digest":"sha256:7ba27a1e9019c76ee5cefa02b12a4b14726745a6a56c06e8a1a65b044d76abc1","observation_id":"10cbd289-f5b3-4176-b54d-eef634840de4","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.07063","last_updated":"2026-05-08T00:16:11Z","snapshot_observed_at":"2026-08-11T07:15:19.934764Z","submitted_at":"2026-05-08T00:16:11Z","title":"Dr. Post-Training: A Data Regularization Perspective on LLM Post-Training","version":1},"reference_index":124,"source":"arxiv_source","source_observed_at":"2026-05-11T01:57:40.347786Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.07063"},"observation_digest":"sha256:9655ca981990cca84a357e4f599d82bc8c3d6db9dd8afefac7d38fa6c9a4e74f","observation_id":"7bd7c218-ff88-40ea-9006-d14104726a17","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.07724","last_updated":"2026-06-03T12:55:17Z","snapshot_observed_at":"2026-07-06T23:20:06.574823Z","submitted_at":"2026-05-08T13:27:23Z","title":"Curated Synthetic Data Doesn't Have to Collapse: A Theoretical Study of Generative Retraining with Pluralistic Preferences","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-11T02:30:14.693348Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.07724"},"observation_digest":"sha256:ae2fb764b0f0c14e4cfc83c9dfdcce26bce625acf76d7e55b1d3d5f40e592bb5","observation_id":"a7882be5-e14b-4d10-9a28-b824a2c66a79","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.08378","last_updated":"2026-05-08T18:36:25Z","snapshot_observed_at":"2026-08-02T18:04:56.183678Z","submitted_at":"2026-05-08T18:36:25Z","title":"Reinforcement Learning for Scalable and Trustworthy Intelligent Systems","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-12T01:47:40.772146Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.08378"},"observation_digest":"sha256:fb557d9a829a0ad04c6d55107d034c0932a15e64da7c7880337393ee855d33ca","observation_id":"a228f44e-35ad-4a0f-88c3-af473d5152d3","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.08556","last_updated":"2026-05-08T23:26:35Z","snapshot_observed_at":"2026-07-06T23:20:47.880233Z","submitted_at":"2026-05-08T23:26:35Z","title":"Can Revealed Preferences Clarify LLM Alignment and Steering?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T02:10:50.298626Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.08556"},"observation_digest":"sha256:a64656ecabb643bfaa59a78c937346e7a97d8c627385c12e949a65f619967844","observation_id":"97b9a3c1-143e-47c4-be16-64e2753cb0cd","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.10937","last_updated":"2026-05-11T17:59:25Z","snapshot_observed_at":"2026-08-11T12:59:30.308378Z","submitted_at":"2026-05-11T17:59:25Z","title":"Power Reinforcement Post-Training of Text-to-Image Models with Super-Linear Advantage Shaping","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-12T03:33:40.994346Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.10937"},"observation_digest":"sha256:8c5d0319a83a3d4fc9166a5f426dc926849d021b761c2bfd878cf1f1a267f1a5","observation_id":"28c1bddf-928e-457b-96f3-c92bf7e3840d","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.11134","last_updated":"2026-05-29T17:16:57Z","snapshot_observed_at":"2026-08-11T09:54:15.520173Z","submitted_at":"2026-05-11T18:41:12Z","title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-13T06:30:51.812541Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.11134"},"observation_digest":"sha256:e3423b696a9a7fc572287cf7c57af3b7d4ccb7748c94a8a3bb2cca8d519d3490","observation_id":"9d69cc25-5418-41e5-97c0-3b07659707fb","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.12809","last_updated":"2026-05-12T23:01:29Z","snapshot_observed_at":"2026-07-06T23:24:27.821980Z","submitted_at":"2026-05-12T23:01:29Z","title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","version":1},"reference_index":215,"source":"arxiv_source","source_observed_at":"2026-05-14T20:17:01.224864Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.12809"},"observation_digest":"sha256:f045fc546a1fa782cc72e6f23ee155cd65ccbaf846adc521df87dd894a6f6c97","observation_id":"91045b5c-6b56-470a-80bc-2d0ff7c0f63f","resolution":{"observed_at":"2026-05-14T22:44:22.602157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.13875","last_updated":"2026-05-08T06:56:35Z","snapshot_observed_at":"2026-08-02T22:27:33.584660Z","submitted_at":"2026-05-08T06:56:35Z","title":"Common-agency Games for Multi-Objective Test-Time Alignment","version":1},"reference_index":184,"source":"arxiv_source","source_observed_at":"2026-05-15T06:14:53.685486Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.13875"},"observation_digest":"sha256:124ded100f49d235db09ba6430bd32a9ad371322aa909b2bc4a6362d0b3f9ef0","observation_id":"57819ffd-b01a-45c3-895e-8b7debcb0fda","resolution":{"observed_at":"2026-05-15T06:15:06.439800Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.15207","last_updated":"2026-07-08T14:36:15Z","snapshot_observed_at":"2026-07-12T17:52:34.996306Z","submitted_at":"2026-05-01T23:42:57Z","title":"TeamTR: Trust-Region Fine-Tuning for Multi-Agent LLM Coordination","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-19T18:01:06.649723Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.15207"},"observation_digest":"sha256:9385b0902a95a30ba5379e112f476bac9e712b1e7637d957c44ea623518ecfe3","observation_id":"ba3bf406-2a29-4790-b9ba-99a65925cc69","resolution":{"observed_at":"2026-05-19T18:02:42.314028Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.16198","last_updated":"2026-05-15T17:13:27Z","snapshot_observed_at":"2026-08-06T00:18:17.955385Z","submitted_at":"2026-05-15T17:13:27Z","title":"Formal Methods Meet LLMs: Auditing, Monitoring, and Intervention for Compliance of Advanced AI Systems","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T17:26:02.346222Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.16198"},"observation_digest":"sha256:dcbdc94eff07391f4f4b8ea76e4f2a900e675c3daa00cf354ff4159db0e0a7c5","observation_id":"9e08c13b-3780-4ef4-b8d5-341dd7ec9d55","resolution":{"observed_at":"2026-05-20T17:28:48.024488Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.16339","last_updated":"2026-05-07T16:48:48Z","snapshot_observed_at":"2026-08-12T12:06:34.294930Z","submitted_at":"2026-05-07T16:48:48Z","title":"Preference Instability in Reward Models: Detection and Mitigation via Sparse Autoencoders","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-20T22:48:54.238767Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.16339"},"observation_digest":"sha256:dac2d844942ec603dfbb706eb249db869627342b8997064eb9d891b5b2c8a7c1","observation_id":"d47a6bbf-8f01-48ce-ab89-8f3b3b24f7a1","resolution":{"observed_at":"2026-05-20T22:49:10.204764Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.16872","last_updated":"2026-05-16T08:24:06Z","snapshot_observed_at":"2026-07-30T06:08:25.380344Z","submitted_at":"2026-05-16T08:24:06Z","title":"Some[Body] Must Receive That Pain for Agent Accountability","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-19T19:46:03.728266Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.16872"},"observation_digest":"sha256:298ce6b30952a06cda59fd07fcfbe32de6162384aeec4f18b9f4f317079ad137","observation_id":"542eb250-0ca5-4b23-b1cc-734e8f95d18e","resolution":{"observed_at":"2026-05-19T19:47:44.480439Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.17458","last_updated":"2026-05-17T14:00:01Z","snapshot_observed_at":"2026-07-06T23:28:26.397778Z","submitted_at":"2026-05-17T14:00:01Z","title":"ClaHF: A Human Feedback-inspired Reinforcement Learning Framework for Improving Classification Tasks","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-20T15:11:27.420642Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.17458"},"observation_digest":"sha256:50ef7f0aaf4716cd824cca5ebac44d5e6baa11a31de7cf0f4e675d153b2aff75","observation_id":"134094df-a128-4114-b62e-dc477c9eadb9","resolution":{"observed_at":"2026-05-20T15:13:24.966375Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.19516","last_updated":"2026-05-19T08:13:12Z","snapshot_observed_at":"2026-07-06T23:30:16.094579Z","submitted_at":"2026-05-19T08:13:12Z","title":"Base Models Look Human To AI Detectors","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-20T06:13:10.815204Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.19516"},"observation_digest":"sha256:79d5effa14939230034cf5281749beb792068a75ef8db41e43400e5e51deb7dd","observation_id":"da158d90-1513-4cff-b6ca-bf2fc83d9b8f","resolution":{"observed_at":"2026-05-20T06:13:22.514511Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.20654","last_updated":"2026-06-03T08:00:52Z","snapshot_observed_at":"2026-08-02T18:56:31.274465Z","submitted_at":"2026-05-20T03:16:15Z","title":"REFLECTOR: Internalizing Step-wise Reflection against Indirect Jailbreak","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-21T06:16:01.040236Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.20654"},"observation_digest":"sha256:08aad89b268b98ae52faabbdb7cf2244e37dc4b3f921f197fa0313aba9fc6c13","observation_id":"ae5723ab-12d5-4757-9601-5cb5fb29229b","resolution":{"observed_at":"2026-05-21T06:19:42.015118Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.21984","last_updated":"2026-05-21T04:34:00Z","snapshot_observed_at":"2026-08-02T23:47:40.908968Z","submitted_at":"2026-05-21T04:34:00Z","title":"Echo: Learning from Experience Data via User-Driven Refinement","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T06:37:34.840129Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.21984"},"observation_digest":"sha256:e013d4c7a6301afbee499f1c46f41978f07e9495e489b5f323ebb1d44204d023","observation_id":"85dbb226-d4be-4a22-8545-5ed867158fa0","resolution":{"observed_at":"2026-05-22T06:41:10.792145Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.24686","last_updated":"2026-05-23T17:48:09Z","snapshot_observed_at":"2026-07-06T23:34:43.210149Z","submitted_at":"2026-05-23T17:48:09Z","title":"Emotional intelligence in large language models is fragmented across perception, cognition, and interaction","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-30T13:16:38.848193Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.24686"},"observation_digest":"sha256:d5175053ead093ea9a8f6fe8958ed9f8de93f9dc5a7084259f46270e3b5ba4fc","observation_id":"b4c8e3d3-9f7a-4a62-8a41-4f51d03fa8c7","resolution":{"observed_at":"2026-06-30T13:24:40.448632Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.25062","last_updated":"2026-05-24T13:13:16Z","snapshot_observed_at":"2026-08-06T17:13:24.667500Z","submitted_at":"2026-05-24T13:13:16Z","title":"Cultivating Machine Intelligence: The OMEGA Shift from Top-Down Optimization to Autopoietic Cognitive Ecologies","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T23:49:06.077148Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.25062"},"observation_digest":"sha256:9056d8a300c6ca9018037c2e1b31282592d0fa3886b2ce4bda52d05f873da8ff","observation_id":"17887a60-a748-4acf-a1e5-5c02444d73f5","resolution":{"observed_at":"2026-06-30T00:04:06.706440Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":"2307.15217","doi":"10.48550/arxiv.2307.15217","metadata_source":"pith","pith_arxiv_id":"2307.15217","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":"cs.AI","work_id":"73fe40c4-d27f-4883-a2f1-52ea228f44fd","year":2023},"citing_paper":{"arxiv_id":"2605.27914","last_updated":"2026-06-09T02:24:01Z","snapshot_observed_at":"2026-07-06T23:37:36.244456Z","submitted_at":"2026-05-27T03:41:11Z","title":"Does Capability Transfer to Subjective Behavior -- and Would Our Instruments Tell Us? A Self-Evolving, Trust-by-Construction Evaluation Paradigm","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T12:54:36.818698Z"},"links":{"cited_paper":"/paper/2307.15217","citing_paper":"/paper/2605.27914"},"observation_digest":"sha256:d8b24986ccd6c96e24fbf4c6847b168e893012a8ce00b8eda875b2a5e4454808","observation_id":"988ba260-04d3-4ee6-a0e9-134ab4555582","resolution":{"observed_at":"2026-06-29T13:03:26.799213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:19.412323+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2307.15217/citation-record","integrity":"/paper/2307.15217/integrity","json":"/paper/2307.15217/citation-record.json","paper":"/paper/2307.15217"},"outbound":[],"paper":{"arxiv_id":"2307.15217","last_updated":"2023-09-11T17:25:24Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-03T19:11:09.671782Z","submitted_at":"2023-07-27T22:29:25Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 100 inbound Pith citation observations for arXiv:2307.15217."}