{"as_of":"2026-08-07T11:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e6326814922f02b5fc3751da3a5be165938b32e07c773e42c7d165e1c45d9092","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":49,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:58.772797Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-30T23:15:07.886084Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2304.06767","last_updated":"2023-12-01T14:28:06Z","snapshot_observed_at":"2026-08-07T04:02:54.504761Z","submitted_at":"2023-04-13T18:22:40Z","title":"RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment","version":4},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-18T00:46:56.664582Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2304.06767"},"observation_digest":"sha256:d6f959d721f7e0ae530854ba9bf0de94762c87e03b0ea9a428a1eae53b876cbc","observation_id":"72976909-ea44-46d9-88ba-833ff05566db","resolution":{"observed_at":"2026-05-18T00:46:56.897089Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2304.12244","last_updated":"2025-05-27T06:49:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-24T16:31:06Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-13T07:28:24.827546Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2304.12244"},"observation_digest":"sha256:413cff9db0c6a7e86bc4aede5f95053419314a22c458ea8fe154e6b32c93c76c","observation_id":"9ec7fa5b-d481-4847-b890-79898516f46a","resolution":{"observed_at":"2026-05-13T07:28:24.991784Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2307.06435","last_updated":"2024-10-17T01:10:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-12T20:01:52Z","title":"A Comprehensive Overview of Large Language Models","version":10},"reference_index":170,"source":"pdf_text","source_observed_at":"2026-05-19T20:28:38.900026Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2307.06435"},"observation_digest":"sha256:8f64205eb57b2c82addc3f44fcb52e39cbe2f463861534a355f5625bec7c438e","observation_id":"a59b01a7-cea2-4b29-b9d2-895df4f38fb0","resolution":{"observed_at":"2026-05-19T20:28:39.546046Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2308.05374","last_updated":"2024-03-21T00:21:14Z","snapshot_observed_at":"2026-08-02T17:09:20.540788Z","submitted_at":"2023-08-10T06:43:44Z","title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T22:30:44.520703Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2308.05374"},"observation_digest":"sha256:3d3fbb2096fd0ee3f88641ac358a800730edf5c7fe3208ae87d13acf53bbe4fc","observation_id":"e61a961f-06fa-44a7-8530-96ed0368ac11","resolution":{"observed_at":"2026-05-17T22:30:44.734358Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-24T03:23:18.827351Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2402.03300"},"observation_digest":"sha256:2c60e10358d7610f9df1f3dd809833db1c75c517edf9bd1bd2779705a5e20584","observation_id":"64edbcdf-4674-4375-b44a-2eb0f48d4618","resolution":{"observed_at":"2026-05-24T03:23:49.616672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2402.13116","last_updated":"2024-10-21T16:22:33Z","snapshot_observed_at":"2026-08-02T17:17:56.296462Z","submitted_at":"2024-02-20T16:17:37Z","title":"A Survey on Knowledge Distillation of Large Language Models","version":4},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-17T23:31:11.213552Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2402.13116"},"observation_digest":"sha256:8220cabf1f8d4593491ff45800c9b62ace32d29df80f7a072862beaaa78ef46e","observation_id":"9d66a7fd-3736-466f-abf7-3f142b5d1427","resolution":{"observed_at":"2026-05-17T23:31:11.531051Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2408.15339","last_updated":"2026-05-07T20:32:48Z","snapshot_observed_at":"2026-08-06T15:22:00.046544Z","submitted_at":"2024-08-27T18:04:07Z","title":"UNA: A Unified Supervised Framework for Efficient LLM Alignment Across Feedback Types","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T21:22:36.970101Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2408.15339"},"observation_digest":"sha256:504963fd0b973b96d99d94319c047239db79472d6d183a1edeb2f0c73441d319","observation_id":"8fa9f457-f4b5-46d5-b3bf-c65a6b2866e0","resolution":{"observed_at":"2026-05-23T21:23:27.482146Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2409.18169","last_updated":"2026-04-23T18:48:49Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:55:22Z","title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","version":6},"reference_index":175,"source":"pdf_text","source_observed_at":"2026-05-23T20:58:16.237327Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2409.18169"},"observation_digest":"sha256:bf6274b5cd2a5dff093962d267c4def068f2b2aedc01d4b31abd92421877b903","observation_id":"b510fc61-533d-4dc4-b764-8299bbf4e1bc","resolution":{"observed_at":"2026-05-23T20:58:25.928533Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-11T07:53:38.715353Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2409.19256"},"observation_digest":"sha256:0d75a2c4695273da85a8027232d1a4c079a801f2727ec1ff096b664104f85f35","observation_id":"baab4df5-0a6d-4668-9373-c572083d1de9","resolution":{"observed_at":"2026-05-11T07:53:38.963721Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"reference_index":200,"source":"pdf_text","source_observed_at":"2026-05-23T17:33:13.394338Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2411.15594"},"observation_digest":"sha256:162ac7c027b24e0c4df1c287e276cfdfa1957026146a8aae3b8a65de3a4265a0","observation_id":"1944b7f2-59b0-450d-b4d3-a13dc4e8c1eb","resolution":{"observed_at":"2026-05-23T17:35:43.900965Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-07T11:25:58.014876Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02573","last_updated":"2025-06-03T07:53:55Z","snapshot_observed_at":"2026-08-07T11:18:57.407955Z","submitted_at":"2025-06-03T07:53:55Z","title":"IndoSafety: Culturally Grounded Safety for LLMs in Indonesian Languages","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T11:25:58.014876Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2506.02573"},"observation_digest":"sha256:2a284bf942f503932f13648f16a5f1151d0eb8d7dd0f22c0365be36582ace091","observation_id":"19f25e95-2fdc-4404-8b5f-2b5dda7d516d","resolution":{"observed_at":"2026-08-07T11:25:58.014876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-07T11:26:58.772797Z","title":"Rrhf: Rank responses to align language mod- els with human feedback without tears.arXiv preprint arXiv:2304.05302,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02698","last_updated":"2025-06-06T03:14:26Z","snapshot_observed_at":"2026-08-07T11:15:42.841737Z","submitted_at":"2025-06-03T09:47:22Z","title":"Smoothed Preference Optimization via ReNoise Inversion for Aligning Diffusion Models with Varied Human Preferences","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:58.772797Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2506.02698"},"observation_digest":"sha256:c6062f7263bae722569c79df29062608cb5dc92cf0963bcd35eb2bed149fbdac","observation_id":"11d9f62a-5dd0-4174-9e11-c933b93f8e55","resolution":{"observed_at":"2026-08-07T11:26:58.772797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-07T05:47:21.717895Z","title":"Rrhf: Rank responses to align language mod- els with human feedback without tears","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07035","last_updated":"2025-06-08T07:59:09Z","snapshot_observed_at":"2026-08-07T05:41:12.179171Z","submitted_at":"2025-06-08T07:59:09Z","title":"AnnoDPO: Protein Functional Annotation Learning with Direct Preference Optimization","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:47:21.717895Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2506.07035"},"observation_digest":"sha256:53fe7c513c186bf52cd052330b1c4d380d3b0a83a05f2b1c6c4f089e2b75b564","observation_id":"eb4a2363-412a-44da-98fa-a8a3cae7000a","resolution":{"observed_at":"2026-08-07T05:47:21.717895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-07T05:48:04.370319Z","title":"Zheng Yuan, Hongyi Yuan, Chuanqi Tan, Wei Wang, Songfang Huang, and Fei Huang","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07165","last_updated":"2025-06-08T14:31:06Z","snapshot_observed_at":"2026-08-07T05:38:06.929531Z","submitted_at":"2025-06-08T14:31:06Z","title":"AMoPO: Adaptive Multi-objective Preference Optimization without Reward Models and Reference Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:48:04.370319Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2506.07165"},"observation_digest":"sha256:a40eff3cff4dfad5503226ae681e28a4ec8c3f1b8491f9f46372c98d97bd11c5","observation_id":"a6c9871d-7ba5-4d27-ba1d-24fa4ab96c53","resolution":{"observed_at":"2026-08-07T05:48:04.370319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-07T00:31:49.283692Z","title":"Rrhf: Rank responses to align language models with human feedback without tears,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.13702","last_updated":"2026-06-01T10:55:41Z","snapshot_observed_at":"2026-08-07T08:32:28.024919Z","submitted_at":"2025-06-16T17:06:27Z","title":"Value-Free Policy Optimization via Reward Partitioning","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:31:49.283692Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2506.13702"},"observation_digest":"sha256:f61ada2ee79a916f08948f3ef7ddda61a4f0993a36be71a62eeb71a26ef9efce","observation_id":"0d3ab8c7-82c6-472a-85b1-493e848b096b","resolution":{"observed_at":"2026-08-07T00:31:49.283692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T23:49:48.654069Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16024","last_updated":"2025-06-19T04:44:34Z","snapshot_observed_at":"2026-08-06T23:42:02.504756Z","submitted_at":"2025-06-19T04:44:34Z","title":"From General to Targeted Rewards: Surpassing GPT-4 in Open-Ended Long-Context Generation","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T23:49:48.654069Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2506.16024"},"observation_digest":"sha256:7d153d79c8cec4872cbe5fb364ff314f5f629b2e71e6ed4aef49c8c3ad6e09b9","observation_id":"a45831ae-3018-49f0-b5f9-0c5dc3b00652","resolution":{"observed_at":"2026-08-06T23:49:48.654069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T21:02:42.520388Z","title":"Rrhf: Rank responses to align language models with human feedback without tears","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01368","last_updated":"2025-07-02T05:10:29Z","snapshot_observed_at":"2026-08-07T10:10:33.545000Z","submitted_at":"2025-07-02T05:10:29Z","title":"Activation Reward Models for Few-Shot Model Alignment","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T21:02:42.520388Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.01368"},"observation_digest":"sha256:9d451e31eebe3b3e5b1b621d805d675d47e0858e395fe9cf7d906b6a9f13b8ec","observation_id":"e8f5b9bb-35ec-4f6f-84c6-d5d8950363da","resolution":{"observed_at":"2026-08-06T21:02:42.520388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T19:29:10.475405Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05617","last_updated":"2025-07-08T02:54:15Z","snapshot_observed_at":"2026-08-07T03:43:31.678279Z","submitted_at":"2025-07-08T02:54:15Z","title":"Flipping Knowledge Distillation: Leveraging Small Models' Expertise to Enhance LLMs in Text Matching","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T19:29:10.475405Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.05617"},"observation_digest":"sha256:71aa5f65a99454987eee692ab3d4ff657870d9da79e3a69bf6a995d2375ad9e3","observation_id":"de9c2c0e-ad7c-44cd-93c5-03e4f816e9f0","resolution":{"observed_at":"2026-08-06T19:29:10.475405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T17:26:45.506333Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10923","last_updated":"2025-07-15T02:30:33Z","snapshot_observed_at":"2026-08-06T17:18:51.361831Z","submitted_at":"2025-07-15T02:30:33Z","title":"Enhancing Safe and Controllable Protein Generation via Knowledge Preference Optimization","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T17:26:45.506333Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.10923"},"observation_digest":"sha256:bb0ec424fc0c3bc7e980c09e7bde511c1682bfdc1a63a1cc5cb999e33acbfb00","observation_id":"f5e67739-421f-4012-a12f-e1cddf0cac87","resolution":{"observed_at":"2026-08-06T17:26:45.506333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T16:34:25.279111Z","title":"Rrhf: Rank responses to align language models with human feedback without tears.arXiv preprint arXiv:2304.05302,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13158","last_updated":"2025-07-17T14:22:24Z","snapshot_observed_at":"2026-08-07T07:16:46.127762Z","submitted_at":"2025-07-17T14:22:24Z","title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","version":1},"reference_index":110,"source":"pdf_text","source_observed_at":"2026-08-06T16:34:25.279111Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.13158"},"observation_digest":"sha256:259b55b66d537826cdff80a956591b127bab52622e9c9d6164d724f025aa9f65","observation_id":"26a849c7-7e62-4a1a-9237-5698fd209cb9","resolution":{"observed_at":"2026-08-06T16:34:25.279111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T14:13:07.226831Z","title":"Rrhf: Rank responses to align language models with human feedback without tears","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19672","last_updated":"2025-07-25T20:52:58Z","snapshot_observed_at":"2026-08-06T15:54:13.981735Z","submitted_at":"2025-07-25T20:52:58Z","title":"Alignment and Safety in Large Language Models: Safety Mechanisms, Training Paradigms, and Emerging Challenges","version":1},"reference_index":260,"source":"arxiv_source","source_observed_at":"2026-08-06T14:13:07.226831Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.19672"},"observation_digest":"sha256:e3f59923af88bab7a81359ba751c83a17c586c79a595a0364fe0bd69c5afc422","observation_id":"48f014e5-7d23-4d3f-bc37-cd9703e6a585","resolution":{"observed_at":"2026-08-06T14:13:07.226831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:5a56f7ae422acf469084b3cc4e8b19ee9ea466229ec63e26fd35467b4cd3714e","observation_id":"a8cc6708-05ee-4ece-a79a-1f80bff6b554","resolution":{"observed_at":"2026-05-14T22:23:15.036632Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T12:09:39.829280Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22037","last_updated":"2025-07-29T17:39:48Z","snapshot_observed_at":"2026-08-06T15:54:16.679483Z","submitted_at":"2025-07-29T17:39:48Z","title":"Secure Tug-of-War (SecTOW): Iterative Defense-Attack Training with Reinforcement Learning for Multimodal Model Security","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T12:09:39.829280Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2507.22037"},"observation_digest":"sha256:c3ca38e9c2264299e18a64b6bc0e25bd3fa0e29c4c16c1e843d064f615a52c65","observation_id":"b5d488ce-ff01-4ef7-b4b2-c88ba28494c0","resolution":{"observed_at":"2026-08-06T12:09:39.829280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-06T04:47:25.428063Z","title":"arXiv preprint arXiv:2304.05302 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.03054","last_updated":"2025-08-05T03:58:15Z","snapshot_observed_at":"2026-08-06T15:30:46.727768Z","submitted_at":"2025-08-05T03:58:15Z","title":"Beyond Surface-Level Detection: Towards Cognitive-Driven Defense Against Jailbreak Attacks via Meta-Operations Reasoning","version":1},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-08-06T04:47:25.428063Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2508.03054"},"observation_digest":"sha256:b19f74104361f79e47647180620a9b0fffdb8e2c692b7c23f1d65e2516dbddee","observation_id":"6735acff-92f5-4e47-a1e8-a09b988f2502","resolution":{"observed_at":"2026-08-06T04:47:25.428063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2508.15119","last_updated":"2026-05-07T17:41:50Z","snapshot_observed_at":"2026-07-06T22:15:50.240911Z","submitted_at":"2025-08-20T23:07:10Z","title":"Flexible Agent Alignment with Goal Inference from Open-Ended Dialog","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T21:35:10.700387Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2508.15119"},"observation_digest":"sha256:e75895a1cfe6d870c70698b0c696357606a0edb40acda11574e91d6f92b7b847","observation_id":"1e311480-4017-468b-932f-7df4c5757651","resolution":{"observed_at":"2026-05-18T21:36:52.365124Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-05T16:30:13.419723Z","title":"Rrhf: Rank responses to align language models with human feedback without tears","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.18391","last_updated":"2025-08-25T18:31:03Z","snapshot_observed_at":"2026-08-05T16:30:13.027254Z","submitted_at":"2025-08-25T18:31:03Z","title":"PKG-DPO: Optimizing Domain-Specific AI systems with Physics Knowledge Graphs and Direct Preference Optimization","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T16:30:13.419723Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2508.18391"},"observation_digest":"sha256:d25c560bf95dcb9afce904833fadfab2e3b4ae599b7c663c184f9f28e3fee943","observation_id":"768289db-0364-4ac0-93cd-24ed24bee1a4","resolution":{"observed_at":"2026-08-05T16:30:13.419723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2509.20265","last_updated":"2026-04-29T12:32:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-24T15:52:36Z","title":"Failure Modes of Maximum Entropy RLHF","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-18T14:02:11.084514Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2509.20265"},"observation_digest":"sha256:a05c700c1df68167aa784a4e800bb45a09b1ff725448e6521b1ed0bf4ac32dee","observation_id":"79211895-3090-4cd1-92ae-5e0fbb5630a0","resolution":{"observed_at":"2026-05-18T14:02:39.687603Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-04T13:15:48.370736Z","title":"Rrhf: Rank responses to align language models with human feedback without tears","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.01167","last_updated":"2026-05-30T06:40:26Z","snapshot_observed_at":"2026-08-04T13:15:39.949629Z","submitted_at":"2025-10-01T17:54:15Z","title":"Simultaneous Multi-objective Alignment Across Verifiable and Non-verifiable Rewards","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-04T13:15:48.370736Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2510.01167"},"observation_digest":"sha256:b0b0bc9f89295555d985cc5dc863f08cde26d5d619e39b796744068c3fa6582a","observation_id":"875c2ac0-b05f-4cb7-9448-571525ed98bb","resolution":{"observed_at":"2026-08-04T13:15:48.370736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2510.17881","last_updated":"2026-04-24T20:38:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-17T23:07:57Z","title":"POPI: Personalizing LLMs via Optimized Natural Language Preference Inference","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-18T05:41:58.231139Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2510.17881"},"observation_digest":"sha256:c168e1838eaa09ce356b349db6c1ef0feacd4f21c445f527bed40d40ea89794e","observation_id":"722fc942-83e3-4dc7-bde2-9b6156d27149","resolution":{"observed_at":"2026-05-18T05:42:24.361109Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2604.09580","last_updated":"2026-02-25T16:01:07Z","snapshot_observed_at":"2026-08-02T18:54:28.820997Z","submitted_at":"2026-02-25T16:01:07Z","title":"OOWM: Structuring Embodied Reasoning and Planning via Object-Oriented Programmatic World Modeling","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-15T19:29:55.075825Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2604.09580"},"observation_digest":"sha256:8ad9b87aded540263175a795d78fb069b6204bd9b9d372eb38719a7bb54fa0fc","observation_id":"30d48663-d659-4a54-bc46-2fd92d99faf2","resolution":{"observed_at":"2026-05-15T19:30:16.442330Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2604.15602","last_updated":"2026-04-17T00:56:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-17T00:56:59Z","title":"GroupDPO: Memory efficient Group-wise Direct Preference Optimization","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-10T09:43:18.432084Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2604.15602"},"observation_digest":"sha256:747dce12b8994da5f870003040d492581c2e27d85b5f334cd41fe43deecc4deb","observation_id":"31089a63-6ed0-4a01-8500-42cf20d59c9d","resolution":{"observed_at":"2026-05-10T09:43:48.777256Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.04539","last_updated":"2026-07-17T05:12:53Z","snapshot_observed_at":"2026-08-02T14:53:21.629442Z","submitted_at":"2026-05-06T06:36:09Z","title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-08T16:57:49.396570Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.04539"},"observation_digest":"sha256:d928a34074f72bb2a86c0d914c3e32689b6861219d7abd5687ba6d25873dd99a","observation_id":"9cfa3750-ac93-497a-965b-151d7f1f4984","resolution":{"observed_at":"2026-05-11T17:56:05.413247Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.04539","last_updated":"2026-07-17T05:12:53Z","snapshot_observed_at":"2026-08-02T14:53:21.629442Z","submitted_at":"2026-05-06T06:36:09Z","title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T03:26:54.426050Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.04539"},"observation_digest":"sha256:e7093fbdd528c77411b5d6a88b716d4aafecb7344f78a1a579d34975c5df9efd","observation_id":"54258816-4d2a-4120-b1c3-b6ea9f080ad3","resolution":{"observed_at":"2026-05-12T07:21:26.380112Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.04539","last_updated":"2026-07-17T05:12:53Z","snapshot_observed_at":"2026-08-02T14:53:21.629442Z","submitted_at":"2026-05-06T06:36:09Z","title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T07:08:39.328446Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.04539"},"observation_digest":"sha256:ecbf92bb4f83f43d4d9690dc19cda5fb7b0d3b367ce64055bfce9ee6c8fbfbc2","observation_id":"7bb9b7cd-b001-41b4-9ec4-06c1ae9352fe","resolution":{"observed_at":"2026-05-13T07:12:28.792093Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-02T14:53:22.597671Z","title":"In Advances in Neural Information Processing Systems (NeurIPS), volume 36, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.04539","last_updated":"2026-07-17T05:12:53Z","snapshot_observed_at":"2026-08-02T14:53:21.629442Z","submitted_at":"2026-05-06T06:36:09Z","title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T14:53:22.597671Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.04539"},"observation_digest":"sha256:afb6e93213b8b0510213f920cd8500c01852ca45e1e30ce667d6dc4a15a9a9f9","observation_id":"985730b5-b7d3-4c31-88ed-4b0322f4ae3e","resolution":{"observed_at":"2026-08-02T14:53:22.597671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.06582","last_updated":"2026-06-21T09:09:33Z","snapshot_observed_at":"2026-07-06T23:19:00.794334Z","submitted_at":"2026-05-07T17:11:22Z","title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-08T12:25:52.847432Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.06582"},"observation_digest":"sha256:413bb9a12dbfa186f70f0e949d91f3bc60df204c882941f879b23dee59251985","observation_id":"9bc0e629-2ba4-4e2f-bfeb-16dbda53b79a","resolution":{"observed_at":"2026-05-11T19:16:08.706755Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.06582","last_updated":"2026-06-21T09:09:33Z","snapshot_observed_at":"2026-07-06T23:19:00.794334Z","submitted_at":"2026-05-07T17:11:22Z","title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-30T23:14:32.494076Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.06582"},"observation_digest":"sha256:b5f75f292dd2cbd3fba2468a65d906efa3ad75817960ae11722dc78235905d18","observation_id":"04d0c94e-1d3a-44d2-8183-29975cecc6ec","resolution":{"observed_at":"2026-06-30T23:15:07.887823Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.06987","last_updated":"2026-05-07T22:05:23Z","snapshot_observed_at":"2026-07-31T05:30:22.249144Z","submitted_at":"2026-05-07T22:05:23Z","title":"Response Time Enhances Alignment with Heterogeneous Preferences","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-11T01:04:26.288913Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.06987"},"observation_digest":"sha256:8a85b74c935faccdb883bb8ac832f08993493c64da42911334ff155f31e26991","observation_id":"96dbc87c-d15d-4ab1-a18a-ff08132f1f16","resolution":{"observed_at":"2026-05-11T04:45:59.770405Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.09433","last_updated":"2026-05-10T09:13:40Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T09:13:40Z","title":"Offline Preference Optimization for Rectified Flow with Noise-Tracked Pairs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-12T02:10:27.595446Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.09433"},"observation_digest":"sha256:f027a5717710a524267f0ffbd276b96f6f587973efecce00ad58f13d496a4638","observation_id":"da909d21-cd01-46ed-9ba2-568e47c4b775","resolution":{"observed_at":"2026-05-12T02:11:15.540831Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.12288","last_updated":"2026-06-10T07:32:21Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:44:33Z","title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","version":1},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-05-13T04:55:55.013900Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.12288"},"observation_digest":"sha256:b0a03b9b47dfe6b6fb5aeca2d67dcc3c489f0cce08fd8ace1f5cfbd85ba012cf","observation_id":"34a8d2c4-5ccb-461c-abae-f3fc672c673d","resolution":{"observed_at":"2026-05-13T04:57:17.252509Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.12288","last_updated":"2026-06-10T07:32:21Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:44:33Z","title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","version":2},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-05-15T05:41:10.714594Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.12288"},"observation_digest":"sha256:a2f8a61ef937b32e9cb8b5252b9405d0ea231acc9bf830c663c750008cf7a772","observation_id":"987a504b-d28b-4e5f-b1fc-d8f7db65f81c","resolution":{"observed_at":"2026-05-15T05:45:06.707485Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2605.23244","last_updated":"2026-05-22T05:25:00Z","snapshot_observed_at":"2026-08-01T14:10:27.190441Z","submitted_at":"2026-05-22T05:25:00Z","title":"Convex Optimization for Alignment and Preference Learning on a Single GPU","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-25T05:01:31.560963Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2605.23244"},"observation_digest":"sha256:a35965f616ac20e8a6223363a3cf1286b572996150aa62f878f52e669e81f3a0","observation_id":"70559131-adfd-4148-8834-7d0046e98dde","resolution":{"observed_at":"2026-05-25T05:06:38.375261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":"2304.05302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-06-30T23:15:07.886084Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":"1212cf79-debd-44aa-a9b3-43c4c99b4db6","year":2023},"citing_paper":{"arxiv_id":"2606.28707","last_updated":"2026-06-27T03:25:53Z","snapshot_observed_at":"2026-07-07T00:02:47.924620Z","submitted_at":"2026-06-27T03:25:53Z","title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-06-30T10:07:39.554999Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2606.28707"},"observation_digest":"sha256:5913392aac1ff369274245a75add57fb2402a902514da464221d29f05e52d98a","observation_id":"79fba758-3eae-47f0-aff4-3efa0a651e42","resolution":{"observed_at":"2026-06-30T12:44:40.106556Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":157,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:9801c8dfcf6aa5cfce5e9d26e0bb0f32d543d2f94fb2844569bf495bbae3009a","observation_id":"70a8db04-0fd0-4fd6-be39-2f830736006c","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-02T08:40:50.066836Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-02T10:24:43.977557Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":158,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:50.066836Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:44834efaa0ddec5940d0f13a3c00225da8da017392341495698d3be67da83538","observation_id":"cc17eec0-cf70-4c10-bc40-2a677e905421","resolution":{"observed_at":"2026-08-02T08:40:50.066836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-07-13T05:26:06.829382Z","title":"Rrhf: Rank responses to align language models with human feedback without tears.arXiv preprint arXiv:2304.05302, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.08968","last_updated":"2026-07-09T22:09:16Z","snapshot_observed_at":"2026-08-02T07:02:01.100420Z","submitted_at":"2026-07-09T22:09:16Z","title":"Every Sample Counts: Supervised Fine-Tuning of Language Models with Pointwise Constraints","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-07-13T05:26:06.829382Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2607.08968"},"observation_digest":"sha256:6401275e288c9b19b5a6779179f4a47bdeed312c4c5c7dfde72174caa8c98a37","observation_id":"98eb984c-4f54-4d6e-aa35-87962f7db8cd","resolution":{"observed_at":"2026-07-13T05:26:06.829382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-01T07:28:27.299214Z","title":"arXiv preprint arXiv:2304.05302 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21453","last_updated":"2026-07-24T02:14:44Z","snapshot_observed_at":"2026-08-03T14:01:50.133228Z","submitted_at":"2026-07-23T15:55:29Z","title":"Test-Time Scaling via Error Localization","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-01T07:28:27.299214Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2607.21453"},"observation_digest":"sha256:38c1e228b45dbcbea35b2847d19db849c8782df44aee97335b4aba7c36e89d22","observation_id":"d3585521-9000-4a22-9543-274e88dc66bb","resolution":{"observed_at":"2026-08-01T07:28:27.299214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-02T13:36:56.283101Z","title":"doi:10.48550/arXiv.2304.05302 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21610","last_updated":"2026-05-20T04:43:02Z","snapshot_observed_at":"2026-08-06T13:09:35.487490Z","submitted_at":"2026-05-20T04:43:02Z","title":"SCOPE and SCION: A Benchmark and an Auditable Reference Pipeline for Schema Induction and Fusion from Text","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-02T13:36:56.283101Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2607.21610"},"observation_digest":"sha256:69f3f93794669343e19520cdff4685624f23ceb105cb6ff4ee624dc0c746204d","observation_id":"da752a54-d624-4528-859d-44f132c7ca38","resolution":{"observed_at":"2026-08-02T13:36:56.283101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.05302","snapshot_observed_at":"2026-08-07T00:14:16.365766Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02713","last_updated":"2026-08-03T17:59:58Z","snapshot_observed_at":"2026-08-07T11:12:19.455451Z","submitted_at":"2026-08-03T17:59:58Z","title":"Quo Vadis, World Modeling?","version":1},"reference_index":211,"source":"pdf_text","source_observed_at":"2026-08-07T00:14:16.365766Z"},"links":{"cited_paper":"/paper/2304.05302","citing_paper":"/paper/2608.02713"},"observation_digest":"sha256:90129463110f09a90b72aa17056d12bfdb1758ca3dccb016d479fcafb1b5ca56","observation_id":"e36ee06a-daf0-477f-a23d-7923d915f246","resolution":{"observed_at":"2026-08-07T00:14:16.365766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.05302/citation-record","integrity":"/paper/2304.05302/integrity","json":"/paper/2304.05302/citation-record.json","paper":"/paper/2304.05302"},"outbound":[],"paper":{"arxiv_id":"2304.05302","last_updated":"2023-10-07T07:01:26Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-04T23:38:14.844517Z","submitted_at":"2023-04-11T15:53:40Z","title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 49 inbound Pith citation observations for arXiv:2304.05302."}