{"as_of":"2026-08-08T00:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bfcd416d04e5b4d02e509fafa9a60f0a1d686d5a82c4d20999769d8e2d62fc8b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:37:50.431221Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T08:07:45.294768Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2401.10020","last_updated":"2025-03-28T00:06:51Z","snapshot_observed_at":"2026-08-07T08:02:34.857823Z","submitted_at":"2024-01-18T14:43:47Z","title":"Self-Rewarding Language Models","version":3},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-05-13T12:01:42.290502Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2401.10020"},"observation_digest":"sha256:2e5e938428117483a0f9e1d37ad1d388866d260a8432b23f3fdc95fed3b0fc1d","observation_id":"be79fbfa-d555-494a-8c4a-1fae5e1094ea","resolution":{"observed_at":"2026-05-13T12:01:42.513253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T14:37:50.431221Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18433","last_updated":"2025-08-13T01:12:42Z","snapshot_observed_at":"2026-08-07T18:05:21.660530Z","submitted_at":"2025-05-24T00:00:43Z","title":"Finite-Time Global Optimality Convergence in Deep Neural Actor-Critic Methods for Decentralized Multi-Agent Reinforcement Learning","version":2},"reference_index":4344,"source":"pdf_text","source_observed_at":"2026-08-07T14:37:50.431221Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.18433"},"observation_digest":"sha256:727416e8ecde9edbad22c53dc233b75c9964ff64a99aeac266e507822597903d","observation_id":"cb628c24-4853-460c-a3d4-056fa4cffffd","resolution":{"observed_at":"2026-08-07T14:37:50.431221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T14:01:06.783481Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.20556","last_updated":"2025-05-26T22:34:42Z","snapshot_observed_at":"2026-08-07T13:49:33.928724Z","submitted_at":"2025-05-26T22:34:42Z","title":"Learning a Pessimistic Reward Model in RLHF","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:01:06.783481Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.20556"},"observation_digest":"sha256:b73ef0cd961ab043d1ce1943bc7c3107ec860602d348597137a6d2fd6c380f85","observation_id":"c925f773-402e-4005-aad0-6c67f4ba00c6","resolution":{"observed_at":"2026-08-07T14:01:06.783481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T14:31:46.467622Z","title":"Gibbs sampling from human feedback: A provable kl-constrained framework for rlhf","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-07T14:26:44.015259Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":133,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:46.467622Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:10d39673737d3d9a66655a2f6667d35f5349f50f304da5930fd4355beaccaf3f","observation_id":"e26b3256-2698-4ae6-8ffa-1f03b099f100","resolution":{"observed_at":"2026-08-07T14:31:46.467622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T13:24:30.488898Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21907","last_updated":"2025-05-31T04:48:02Z","snapshot_observed_at":"2026-08-07T13:17:19.145607Z","submitted_at":"2025-05-28T02:52:39Z","title":"Modeling and Optimizing User Preferences in AI Copilots: A Comprehensive Survey and Taxonomy","version":2},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-07T13:24:30.488898Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.21907"},"observation_digest":"sha256:472647a99575d16c09e0e97f1c12a81c2cd0a41ef8ef1f6e0f155d5c17b54327","observation_id":"2fd6201e-da41-4d52-bed5-5e89a29b156c","resolution":{"observed_at":"2026-08-07T13:24:30.488898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T12:43:49.435213Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint.arXiv e-prints, page arXiv:2312.11456, December 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23927","last_updated":"2025-05-29T18:22:02Z","snapshot_observed_at":"2026-08-07T12:35:50.981783Z","submitted_at":"2025-05-29T18:22:02Z","title":"Thompson Sampling in Online RLHF with General Function Approximation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:43:49.435213Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2505.23927"},"observation_digest":"sha256:e1053ef039de9bde3922c6a72888c44f96ec29a60f10e421fa0969ed3c5f4c3d","observation_id":"d633d1f9-2b01-4aa8-929c-e316d7ca4f71","resolution":{"observed_at":"2026-08-07T12:43:49.435213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T10:50:52.162823Z","title":"InForty-first International Conference on Machine Learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04463","last_updated":"2025-06-04T21:29:11Z","snapshot_observed_at":"2026-08-07T10:39:22.199403Z","submitted_at":"2025-06-04T21:29:11Z","title":"Aligning Large Language Models with Implicit Preferences from User-Generated Content","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:50:52.162823Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2506.04463"},"observation_digest":"sha256:eeaa50fdd0c6d94dbe2d62ba9d32ae87a69cccd9d779e5a3fe1445e1ea130e55","observation_id":"738674b3-804c-466f-8cb8-2176cf94dc03","resolution":{"observed_at":"2026-08-07T10:50:52.162823Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-07T05:51:30.736946Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.06923","last_updated":"2025-06-07T21:23:00Z","snapshot_observed_at":"2026-08-07T12:11:21.205285Z","submitted_at":"2025-06-07T21:23:00Z","title":"Boosting LLM Reasoning via Spontaneous Self-Correction","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:51:30.736946Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2506.06923"},"observation_digest":"sha256:b92d509ba85cfc838ce8d6317157e40218f03946e62b081beecde0ffdfc71abf","observation_id":"2cbcdcb0-37db-4fba-b52b-61e85f8acf78","resolution":{"observed_at":"2026-08-07T05:51:30.736946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-06T22:28:09.397544Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21495","last_updated":"2025-06-26T17:25:49Z","snapshot_observed_at":"2026-08-06T22:21:21.361920Z","submitted_at":"2025-06-26T17:25:49Z","title":"Bridging Offline and Online Reinforcement Learning for LLMs","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T22:28:09.397544Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2506.21495"},"observation_digest":"sha256:821de703da81dd838bdfef3de3fed0f43c11d40df120f0b389253b61d2f06908","observation_id":"aadb1a06-6ba7-44f1-8ad0-d2b140535341","resolution":{"observed_at":"2026-08-06T22:28:09.397544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2507.06419","last_updated":"2026-06-04T20:44:16Z","snapshot_observed_at":"2026-08-06T19:02:44.982053Z","submitted_at":"2025-07-08T21:56:33Z","title":"Teach a Reward Model to Correct Itself: Reward Guided Adversarial Failure Discovery for Robust Reward Modeling","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-19T05:16:22.274580Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2507.06419"},"observation_digest":"sha256:54aaaf19633d930dc6fd0750180fe2acde5aa0f19f34a62bde6edb87dd292c61","observation_id":"7a5f7fda-8506-4f53-83f0-03e83d0c13f5","resolution":{"observed_at":"2026-05-19T05:17:05.794667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-06T16:34:25.249776Z","title":"Gibbs sampling from human feedback: A provable kl-constrained framework for rlhf.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13158","last_updated":"2025-07-17T14:22:24Z","snapshot_observed_at":"2026-08-07T07:16:46.127762Z","submitted_at":"2025-07-17T14:22:24Z","title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-06T16:34:25.249776Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2507.13158"},"observation_digest":"sha256:9d17b7fa5966368a59bf635b16471f77b329afa306f8c93074b370061662d1d5","observation_id":"caa7a167-a786-40b8-ad11-8e33e6c466dd","resolution":{"observed_at":"2026-08-06T16:34:25.249776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2508.16771","last_updated":"2026-07-02T23:23:00Z","snapshot_observed_at":"2026-08-05T17:05:43.834611Z","submitted_at":"2025-08-22T20:08:09Z","title":"EyeMulator: Improving Code Language Models by Mimicking Human Visual Attention","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-18T20:54:30.449792Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2508.16771"},"observation_digest":"sha256:66b815253ea6a077363c5e6595d419db6c5f941aad04923cb3849dba46ce58a8","observation_id":"fce8968b-b7ec-4828-97f6-67d3d1edf231","resolution":{"observed_at":"2026-05-18T20:56:51.807191Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2508.20697","last_updated":"2026-08-04T09:31:07Z","snapshot_observed_at":"2026-08-07T23:09:05.524726Z","submitted_at":"2025-08-28T12:07:11Z","title":"Token Buncher: Shielding LLMs from Harmful Reinforcement Learning Fine-Tuning","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-18T20:40:44.496392Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2508.20697"},"observation_digest":"sha256:ab722ce98efe4313f872f843cb7d55efd00fa9f2719de34b1457c7ea7f5ec87f","observation_id":"e9f47699-f286-41a3-83c0-869043ed15e1","resolution":{"observed_at":"2026-05-18T20:41:50.556600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-04T22:59:14.577593Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.06941","last_updated":"2025-09-08T17:52:56Z","snapshot_observed_at":"2026-08-07T12:11:53.268632Z","submitted_at":"2025-09-08T17:52:56Z","title":"Outcome-based Exploration for LLM Reasoning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T22:59:14.577593Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.06941"},"observation_digest":"sha256:67a4eea17f4d9c2b9fa05bb304d13c2917c1fe4b2b3bf2ade809e6cfb5c1fe42","observation_id":"ca620d9e-6f09-4fc7-bfbe-5a69903cbd8f","resolution":{"observed_at":"2026-08-04T22:59:14.577593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-04T16:07:44.508870Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":207,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:44.508870Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:b416f8f3e1ebb875eb0e7d9da7b22c7bd079db647ba06df8fa51a717c4ad6e1b","observation_id":"f46296f8-9f42-4f61-bc8a-4a54339c294d","resolution":{"observed_at":"2026-08-04T16:07:44.508870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-04T14:57:00.278197Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.22992","last_updated":"2026-08-03T08:38:00Z","snapshot_observed_at":"2026-08-07T00:55:36.865093Z","submitted_at":"2025-09-26T23:08:03Z","title":"T-TAMER: Provably Taming Trade-offs in ML Serving","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-04T14:57:00.278197Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.22992"},"observation_digest":"sha256:fa57bdaf223f3ba931ea23fde301a1685974a9c048e41caef9e888450cb47788","observation_id":"54e2a028-9ce8-437c-82c9-b871889ae3d1","resolution":{"observed_at":"2026-08-04T14:57:00.278197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2509.23102","last_updated":"2026-04-06T19:14:12Z","snapshot_observed_at":"2026-08-03T06:33:00.420621Z","submitted_at":"2025-09-27T04:18:33Z","title":"Multiplayer Nash Preference Optimization","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-18T13:09:54.433720Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2509.23102"},"observation_digest":"sha256:2775e666f5cfda0070f267f1fa35ba8c7c086cce755de9826e39fa282a2fbacb","observation_id":"a187e3c4-ed2a-4fd9-bc9c-25ad2c792f07","resolution":{"observed_at":"2026-05-18T13:11:24.027237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-03T13:42:06.121305Z","title":"Private reinforcement learning with pac and regret guarantees","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2512.23816","last_updated":"2026-06-25T06:56:14Z","snapshot_observed_at":"2026-08-06T05:05:37.952259Z","submitted_at":"2025-12-29T19:20:35Z","title":"Improved Bounds for Private and Robust Alignment","version":2},"reference_index":2000,"source":"pdf_text","source_observed_at":"2026-08-03T13:42:06.121305Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2512.23816"},"observation_digest":"sha256:fde157ca1a84bc353f713ad6232af5488b48f209c31050bc3bda53255f818fcb","observation_id":"29f80c5d-e9d3-4b69-a6ca-ff32d85aa70e","resolution":{"observed_at":"2026-08-03T13:42:06.121305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-03T03:33:41.469348Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-21T13:13:13.293921Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:865b3f336610ded296bc28b5e9d7aa6e3e48a9100e37d2e48640b027d069c5e3","observation_id":"9f384787-e9db-4a2a-bf85-b581734215fd","resolution":{"observed_at":"2026-05-21T13:14:11.017458Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-03T03:33:45.088960Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-03T03:33:41.469348Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T03:33:45.088960Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:38861e5d61a232ab42fb2e34bcf3e97e024024c127468d1993e6b1b49dd387c0","observation_id":"51151045-dabe-41db-85a2-582cede2776b","resolution":{"observed_at":"2026-08-03T03:33:45.088960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2604.13598","last_updated":"2026-04-15T08:08:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-15T08:08:06Z","title":"Enhancing Reinforcement Learning for Radiology Report Generation with Evidence-aware Rewards and Self-correcting Preference Learning","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-10T14:09:06.139625Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2604.13598"},"observation_digest":"sha256:374eba6f2c46acc8f67b3efe617f425da781f9b5fa74fd04c1d1fe4006991670","observation_id":"11128f59-79a4-4dc7-90d6-57e1d37c1fe3","resolution":{"observed_at":"2026-05-10T14:10:28.664074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2604.20933","last_updated":"2026-04-22T11:52:21Z","snapshot_observed_at":"2026-07-30T07:55:19.034385Z","submitted_at":"2026-04-22T11:52:21Z","title":"IRIS: Interpolative R\\'enyi Iterative Self-play for Large Language Model Fine-Tuning","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-10T01:15:12.985803Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2604.20933"},"observation_digest":"sha256:853860d4750b3b5215f4d9c113a2ae9e3261003c84fd9a7772ae916f1a787422","observation_id":"40f61a43-74ac-4b71-8338-b529d1583785","resolution":{"observed_at":"2026-05-11T13:41:05.005307Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2605.01831","last_updated":"2026-05-03T11:45:08Z","snapshot_observed_at":"2026-07-06T23:15:01.968194Z","submitted_at":"2026-05-03T11:45:08Z","title":"RMGAP: Benchmarking the Generalization of Reward Models across Diverse Preferences","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-09T17:27:05.533755Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2605.01831"},"observation_digest":"sha256:780d8c71cce9bc1d025a2488f4d278ef01aee0ce24a2d4447d89f79a62a9796a","observation_id":"110138e7-d015-4b0d-a398-f9f63980401c","resolution":{"observed_at":"2026-05-11T16:21:07.087022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2605.06977","last_updated":"2026-05-07T21:48:26Z","snapshot_observed_at":"2026-07-06T23:19:25.538514Z","submitted_at":"2026-05-07T21:48:26Z","title":"$f$-Divergence Regularized RLHF: Two Tales of Sampling and Unified Analyses","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T01:13:29.292351Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2605.06977"},"observation_digest":"sha256:c2c4653ac26595f078efef94e158aab2413bc9cdba916c09fb1ff3482c58db4c","observation_id":"3ef6e794-1ad0-4748-a798-b34efe831837","resolution":{"observed_at":"2026-05-11T04:35:58.665306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2605.11134","last_updated":"2026-05-29T17:16:57Z","snapshot_observed_at":"2026-08-01T16:30:31.998321Z","submitted_at":"2026-05-11T18:41:12Z","title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-13T06:30:51.812541Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2605.11134"},"observation_digest":"sha256:90d32ea57b10320a74868654311eec7e3dfc9288280044b3fa9f94f7ba590158","observation_id":"4ba08c5f-0a38-41ec-8355-2b6b0715ab24","resolution":{"observed_at":"2026-05-13T06:32:24.270894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2606.03131","last_updated":"2026-06-02T04:18:08Z","snapshot_observed_at":"2026-08-02T20:05:30.693121Z","submitted_at":"2026-06-02T04:18:08Z","title":"HARVE: Hacking-Aware Reward-Head Vector Editing for Robust Reward Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T11:32:16.166724Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2606.03131"},"observation_digest":"sha256:fa14812a8235fb737b22bf4507fa9d5878387c549a5385f5acabc09ae9021da8","observation_id":"eb3dd61e-1eb5-4ff0-841d-65edf791a5a6","resolution":{"observed_at":"2026-07-02T01:46:26.783878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2606.11437","last_updated":"2026-06-09T20:48:48Z","snapshot_observed_at":"2026-08-04T03:39:21.234561Z","submitted_at":"2026-06-09T20:48:48Z","title":"The Power of Test-Time Training for Approximate Sampling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T11:07:47.543592Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2606.11437"},"observation_digest":"sha256:142647914bd884be9324fdd5dbc5483f6e4e04a14aabd3572670f8398d7bc987","observation_id":"07b58e96-e25d-472f-80e9-06479c7bbd52","resolution":{"observed_at":"2026-07-03T08:07:45.296622Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":"2312.11456","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-07-03T08:07:45.294768Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456","venue":null,"work_id":"a7265aa9-5ce8-4a96-acfd-3f069003f21d","year":2024},"citing_paper":{"arxiv_id":"2606.30445","last_updated":"2026-06-29T15:17:42Z","snapshot_observed_at":"2026-08-07T23:00:10.618237Z","submitted_at":"2026-06-29T15:17:42Z","title":"When Does Online Imitation Learning Help in LLM Post-Training? The Role of (Non-)Realizability Beyond Horizon","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-30T07:41:39.266071Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2606.30445"},"observation_digest":"sha256:449ce1069d5215b469d64b2b446a16a89ba72ae65039cdf0e88494f89d6f4036","observation_id":"cec4c83d-8ba5-45b8-b569-d375ceb5c8f1","resolution":{"observed_at":"2026-06-30T07:44:21.719763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-01T23:59:41.739998Z","title":"mlr.press/v216/wimmer23a.html","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15196","last_updated":"2026-08-04T09:02:09Z","snapshot_observed_at":"2026-08-07T23:09:33.362486Z","submitted_at":"2026-07-16T16:52:54Z","title":"Subjective Risk Decomposition: A New View for Uncertainty Quantification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T23:59:41.739998Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2607.15196"},"observation_digest":"sha256:27e6bee7208b1131997f0075bd0f8b3086ae7978c18e609677afdc0ba542ca80","observation_id":"05b50ba8-3813-405f-8765-7a6df5cff1c2","resolution":{"observed_at":"2026-08-01T23:59:41.739998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11456","snapshot_observed_at":"2026-08-02T10:01:59.330889Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl-constraint.arXiv preprint arXiv:2312.11456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16240","last_updated":"2026-06-26T03:03:15Z","snapshot_observed_at":"2026-08-08T00:34:09.048494Z","submitted_at":"2026-06-26T03:03:15Z","title":"Normalized Rewards for Preference Optimization","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T10:01:59.330889Z"},"links":{"cited_paper":"/paper/2312.11456","citing_paper":"/paper/2607.16240"},"observation_digest":"sha256:17f4f6de4a58061d9d59a4ab0e0163e7b7898fad7ccf967a458a765f0ac93e01","observation_id":"ab20eedc-4fd4-4669-abb6-94f61bdc20e0","resolution":{"observed_at":"2026-08-02T10:01:59.330889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.11456/citation-record","integrity":"/paper/2312.11456/integrity","json":"/paper/2312.11456/citation-record.json","paper":"/paper/2312.11456"},"outbound":[],"paper":{"arxiv_id":"2312.11456","last_updated":"2024-05-01T14:50:56Z","latest_version":4,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-02T03:59:22.576409Z","submitted_at":"2023-12-18T18:58:42Z","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2312.11456."}