{"as_of":"2026-08-07T23:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7ae8e1826b3958a68b7b2ec72d669a40e081ddeb644a2090b027a8d834550fa7","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:19:36.208057Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.05748/citation-record","integrity":"/paper/2506.05748/integrity","json":"/paper/2506.05748/citation-record.json","paper":"/paper/2506.05748"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:41.094546Z","title":"Each category of algorithms presents unique benefits and constraints, rendering their integrated application beneficial in real -world scenarios","venue":null,"work_id":"2748334d-5e30-4865-96dc-b64cf211c62c","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:32.904699Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:257e97065a7d308ba9b1587a27d1f635f27f9cb7bda67982f4a6031662853768","observation_id":"bfc05f2a-2cd1-423e-b517-028202c5982b","resolution":{"observed_at":"2026-08-07T10:19:41.116682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:41.052485Z","title":"LLM-as- a-Judge","venue":null,"work_id":"89f6d4e1-5412-493d-8142-180185779c60","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:32.952718Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:3fc9bef80815cbbc542d71fc9f3fcd54546a892ff743856c21c44bf42d27508b","observation_id":"4bf31566-4f24-4634-8570-42b5186bb4d4","resolution":{"observed_at":"2026-08-07T10:19:41.070207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.828779Z","title":"score\" field in [-1, 1] and a short","venue":null,"work_id":"a63fccda-ba97-4375-a7d9-9473386417b5","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.128347Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:c148581526120b5a24d991703494be62e53d95f23353e2d179c9cf10eadec254","observation_id":"c4e8b6e8-0188-4312-a89d-4c30f929e8c7","resolution":{"observed_at":"2026-08-07T10:19:40.876633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.574827Z","title":"𝑏𝑒𝑡𝑡𝑒𝑟\":","venue":null,"work_id":"b0cdd5f6-83a0-4745-9ca8-e099c22cf288","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.304988Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:2c36a829e39336797201885515588ad2e59a8d21b32d1556b55481119ee56dba","observation_id":"9c517a11-173d-4123-849e-a78414a33087","resolution":{"observed_at":"2026-08-07T10:19:40.636409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.686982Z","title":"be funnier","venue":null,"work_id":"e8cccb6c-e444-4fd8-bb68-fd20df5f9d90","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.229860Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:cd7fcb3c669d72bd0b1ecc533dbebcc8fa03036dd597cf9beaaba4b2d8464288","observation_id":"0394a0ad-67b0-4824-8573-f5df187a27c4","resolution":{"observed_at":"2026-08-07T10:19:40.765243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.337606Z","title":"plug -and-play","venue":null,"work_id":"bf3826be-504e-4c39-809c-960d49d29dd8","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.434466Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:99092894fec41da38261ffa10ae59d68c7395035d92210038e9ea6f7270c3f5f","observation_id":"21804ec0-a5c9-4bad-a29b-1e5668030674","resolution":{"observed_at":"2026-08-07T10:19:40.366764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.443151Z","title":"Which answer is better? Return ‘A’ or ‘B’ only","venue":null,"work_id":"83183cee-ac06-4508-a0c8-522878a18241","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.363323Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:cd9bb5de9d73375cbaf260462aca37cf6bbd0c7d95425ff76d8767d8b03d6964","observation_id":"ed1ee598-6ad1-4a86-b403-bab2b8c2cefc","resolution":{"observed_at":"2026-08-07T10:19:40.499258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.824666Z","title":null,"venue":null,"work_id":"37dab451-4c59-4f9f-acad-d2418ee6a9f2","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.586566Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:0eca54f4b40f5278ca36f6628943da8a6b92e59009818f7534accb65244b7cc9","observation_id":"2966fd8d-7156-4a75-887e-f9bfc63802e3","resolution":{"observed_at":"2026-08-07T10:19:39.929275Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.059753Z","title":"A” or “B","venue":null,"work_id":"32d8cba5-112a-4c85-a270-d34cc1b95863","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.507669Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:7ab365642a46f7ac5dadfee4cba901949cb930a54bfe4a8f09c8365fb115b479","observation_id":"07f5747b-7a14-49e1-9a5b-c3a911624fd7","resolution":{"observed_at":"2026-08-07T10:19:40.160618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4287.34542","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.443445Z","title":"Online and Offline Reinforcement Learning by Planning with a Learned Model,","venue":null,"work_id":"2969fcd4-5738-480e-af7e-41c0e7af41d0","year":2021},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.462170Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:e07e57c7368e0a81ec06d122b0e8769e5b691dacccbb867580397199334e4b68","observation_id":"008d4bbd-9253-4423-838d-373d19c01c03","resolution":{"observed_at":"2026-08-07T10:19:37.491599Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.505765Z","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model Oral,","venue":null,"work_id":"91e86425-f9a5-47c1-affe-b7f7cf0dea0e","year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.659848Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:c93e5e2e80d552abf562e4ed9d637c9559aa64e8b9662bcc91e386a4843a2a42","observation_id":"8bb93105-9cad-4e16-aae9-7de035cd3255","resolution":{"observed_at":"2026-08-07T10:19:39.631521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2312.14925","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:36.917545Z","title":"A Survey of Reinforcement Learning from Human Feedback,","venue":null,"work_id":"133ebc6d-ab80-4282-949f-5b64dcbc1c67","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.727935Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:f2e27e7fdd5939397e53ecd31602e9cc70e6afcb61dfc24ca2e5e125afb2989c","observation_id":"4230bc2a-447e-413d-b868-637db43f9f7d","resolution":{"observed_at":"2026-08-07T10:19:36.981902Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-03T22:08:26.692731+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T22:08:26.692731+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T10:19:33.848792Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.848792Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:815e60ac33ee0af4f788f56dd643b864d874337c13f8336ad4f0ac86a40d47fc","observation_id":"52201413-e62d-4376-99d4-f540ac16dbc2","resolution":{"observed_at":"2026-08-07T10:19:33.848792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:33.926564Z","title":"Security and Privacy Challenges of Large Language Models: A Survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.926564Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:054465cf808601a3e4c82fcd80c5276dc44bd60f491f327a309e52aca2807d8e","observation_id":"650048c6-1110-4069-93d1-b38e3de8259a","resolution":{"observed_at":"2026-08-07T10:19:33.926564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:34.046186Z","title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.046186Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:6a516ba8d4870efe93bc398822c313544f6fb0554bc588f19da1339101e0a691","observation_id":"9f9e4aee-6743-4532-8a83-70b6c4ab6cb9","resolution":{"observed_at":"2026-08-07T10:19:34.046186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T10:19:34.135241Z","title":"Qwen2.5 Technical Report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.135241Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:54623994d062a83aa231d7d42e16f86f4707129921ac37b7fce9fee1d7e4f698","observation_id":"9a466f73-917c-48b2-9f60-e20479759f8d","resolution":{"observed_at":"2026-08-07T10:19:34.135241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.258894Z","title":"Self-rewarding language models,","venue":null,"work_id":"9b36bab2-5e64-4913-995a-fba057e96197","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.252515Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:3571c5d252464a1f4a048ecfaa187ae0c9b53e63df76097b3751131d60e1b88b","observation_id":"008311f0-f88e-4e7a-a34e-34c079fed166","resolution":{"observed_at":"2026-08-07T10:19:39.400464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:39.014875Z","title":"RLAIF: Scaling Reinforcement Learning from Human Feedback with AI Feedback,","venue":null,"work_id":"68d849ae-d718-4cbe-a5c9-3188122f2161","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.294065Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:2601c81bb7b9550ee3180ce01b9ef9bb9f9f86a8ff4939a41e4a01b438da606b","observation_id":"d4f62995-bf80-46ad-bb75-e2a1664aeb75","resolution":{"observed_at":"2026-08-07T10:19:39.116623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:34.415108Z","title":"Evaluating Text -to-Visual Generation with Image -to-Text Generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.415108Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:03ac90bbde812f5b30072ab1cd4d35542fd83bb5ada98049348286ebe7c2db78","observation_id":"3cdbe7f3-4c87-41ac-8be6-f70283a8170b","resolution":{"observed_at":"2026-08-07T10:19:34.415108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:40.959168Z","title":"more proficient","venue":null,"work_id":"8b29270b-1f5a-4763-9e3a-6429ee71deaa","year":null},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:33.043123Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:e5258a0d23ae11c48c0ed8b9d9cbc16130d00e7c0795b76798ff028d5f7aa843","observation_id":"2e689453-8921-4ec3-85a3-8d38b9adadda","resolution":{"observed_at":"2026-08-07T10:19:41.034025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0270.36022","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.243015Z","title":"Training Language Models to Follow Instructions with Human Feedback,","venue":null,"work_id":"ff542819-e434-4c4e-9c5d-4506384430fb","year":2022},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.523346Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:9f033d4d9912499f1be0be46703c7115346b43bea33d1f1f3d2c4c0e2eb2cc76","observation_id":"50ef29fd-e9e9-4a45-bdb5-a57243e42712","resolution":{"observed_at":"2026-08-07T10:19:37.289842Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10400","last_updated":"2025-02-24T08:57:10Z","snapshot_observed_at":"2026-08-07T21:53:28.691375Z","submitted_at":"2024-12-05T16:10:42Z","title":"Reinforcement Learning Enhanced LLMs: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10400","snapshot_observed_at":"2026-08-07T10:19:34.611764Z","title":"Reinforcement Learning Enhanced LLMs: A Survey,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.611764Z"},"links":{"cited_paper":"/paper/2412.10400","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:22f0318ce1c9520b06b409583ce5b40e223414e5396d8559d73dc67913cdc7f6","observation_id":"b3f198b2-d25f-4a79-afe5-066bbddecdb4","resolution":{"observed_at":"2026-08-07T10:19:34.611764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:34.727234Z","title":"Survey on Large Language Model -Enhanced Reinforcement Learning: Concept, Taxonomy, and Methods,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.727234Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:8ca74c285e2f109c6275911cf22d4d573b6488d1d8ee2d9cd666bc9f64db7c27","observation_id":"27d2466d-933e-4b88-ad92-f2b3074f42d3","resolution":{"observed_at":"2026-08-07T10:19:34.727234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.17376","last_updated":"2025-04-24T08:50:01Z","snapshot_observed_at":"2026-08-07T15:59:36.272840Z","submitted_at":"2025-04-24T08:50:01Z","title":"On-Device Qwen2.5: Efficient LLM Inference with Model Compression and Hardware Acceleration","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.17376","snapshot_observed_at":"2026-08-07T10:19:34.793339Z","title":"On-Device Qwen2.5: Efficient LLM Inference with Model Compression and Hardware Acceleration,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.793339Z"},"links":{"cited_paper":"/paper/2504.17376","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:800b49ebe1168208163abc24b96d409db14f1441c374a443c47ae49e4479ac78","observation_id":"74ae2aea-c188-4db3-8f8c-e0aa6aeca16a","resolution":{"observed_at":"2026-08-07T10:19:34.793339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.02554","last_updated":"2023-04-05T16:17:32Z","snapshot_observed_at":"2026-08-06T05:29:37.931917Z","submitted_at":"2023-04-05T16:17:32Z","title":"Human-like Summarization Evaluation with ChatGPT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.02554","snapshot_observed_at":"2026-08-07T10:19:34.827221Z","title":"Human -like Summarization Evaluation with ChatGPT,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.827221Z"},"links":{"cited_paper":"/paper/2304.02554","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:6a90482731ef6f149296274f581b74ceb4571fee3782e2048d9e2f15488e9752","observation_id":"7ed344d8-cc99-4521-8a1f-236530c45dda","resolution":{"observed_at":"2026-08-07T10:19:34.827221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02736","last_updated":"2024-10-04T03:57:47Z","snapshot_observed_at":"2026-08-01T08:21:19.528254Z","submitted_at":"2024-10-03T17:53:30Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02736","snapshot_observed_at":"2026-08-07T10:19:34.917509Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.917509Z"},"links":{"cited_paper":"/paper/2410.02736","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:743365b485dee1f4def5657d6493f63c4776137ace90978444ebf8a6ec0a95cb","observation_id":"2d96a1e9-2f2a-4270-9eba-9578b9740e94","resolution":{"observed_at":"2026-08-07T10:19:34.917509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.11239","last_updated":"2024-10-02T11:07:14Z","snapshot_observed_at":"2026-08-03T08:44:22.056144Z","submitted_at":"2024-09-17T14:40:02Z","title":"LLM-as-a-Judge & Reward Model: What They Can and Cannot Do","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.11239","snapshot_observed_at":"2026-08-07T10:19:34.977810Z","title":"LLM-as-a-Judge & Reward Model: What They Can and Cannot Do,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:34.977810Z"},"links":{"cited_paper":"/paper/2409.11239","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:73bc24b8ad40290428827fe7ee044e90cc1decaaa795b3c756f71a2211fb16a0","observation_id":"be93dd90-2e75-4460-9faa-018aaa2009e0","resolution":{"observed_at":"2026-08-07T10:19:34.977810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.789565Z","title":"RLAIF vs. RLHF: scaling reinforcement learning from human feedback with AI feedback,","venue":null,"work_id":"138a76fd-034e-4b0d-8986-95656d8113f4","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.034720Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:90a368f1366e0a176d54f14003b2e969bc3f067fe1c58ca1bc4b5de1b1ad394e","observation_id":"b0a10ede-dbbe-42c8-b22f-a4d359ab22be","resolution":{"observed_at":"2026-08-07T10:19:38.869784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:35.096249Z","title":"Large Language Models Can Self -Improve,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.096249Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:06a4bbc19068a6a3890b6cb928fbecc7fdb575bb5e260becb3094ca03d617509","observation_id":"fd153a62-f658-4573-8f88-c7e1ac52c102","resolution":{"observed_at":"2026-08-07T10:19:35.096249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.612071Z","title":"Advancing Large Language Model Attribution through Self-Improving,","venue":null,"work_id":"1f178dff-abcb-45ac-8293-1c0745091597","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.180761Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:f07df2c26ed01a5b12941b5b1e17a9834028d0b33158d23ec3e932bd9b21d5cf","observation_id":"8bac4ca2-9414-4a58-9b45-926eb8f81762","resolution":{"observed_at":"2026-08-07T10:19:38.691907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13006","last_updated":"2025-03-30T17:59:47Z","snapshot_observed_at":"2026-07-06T19:05:05.530925Z","submitted_at":"2024-08-23T11:49:01Z","title":"Systematic Evaluation of LLM-as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13006","snapshot_observed_at":"2026-08-07T10:19:35.386305Z","title":"Systematic Evaluation of LLM -as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.386305Z"},"links":{"cited_paper":"/paper/2408.13006","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:cad7bac6d6ef32674a6b888a336a0973539c440f41a8940e026751fb739c60a8","observation_id":"7c9c9731-28a4-45d0-914a-91332507933e","resolution":{"observed_at":"2026-08-07T10:19:35.386305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:35.466565Z","title":"Can LLM be a Personalized Judge?,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.466565Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:635fb177e055fc9dc50f65a213b1f5a51c24ac1775e6763b4cd17685522cbca0","observation_id":"3ed7257e-2d9d-42ea-b02f-56b633518d3b","resolution":{"observed_at":"2026-08-07T10:19:35.466565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.418751Z","title":"ReST-MCTS*: LLM Self- Training via Process Reward Guided Tree Search,","venue":null,"work_id":"3b3afa43-3a3e-4c84-af76-13f13d1966f6","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.548712Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:bc2c425523470367f03ac82d604a9b8b9c2229fdc1e7ee77a3c183a7853b7f70","observation_id":"485880b4-b338-414d-b015-92fa35054ec6","resolution":{"observed_at":"2026-08-07T10:19:38.522228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:38.126489Z","title":"Self-Play Preference Optimization for Language Model Alignment,","venue":null,"work_id":"f8ef14e2-593e-4225-b94d-8189bf0f7cce","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.634935Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:1f6d2e869eb96a147a9191d829f945ac8312d258832ee0637c34a6bf1b3dad9d","observation_id":"56da1a36-06d1-4eee-a34c-2ce65173a5a5","resolution":{"observed_at":"2026-08-07T10:19:38.280897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.666571Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":"8a149f52-5c30-44d9-9978-3397db30c6dd","year":2022},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.826010Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:1ff1c3b3449716f3c9dce03e990db964fd00ccb27741970624b6053324b2e11f","observation_id":"599d70aa-640b-4ce5-836e-8eaff13c67e2","resolution":{"observed_at":"2026-08-07T10:19:37.785037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-07T10:19:35.928792Z","title":"Constitutional AI: Harmlessness from AI Feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.928792Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:0683d151839519e409df05fed5c3bb07b43b108e9006ee50494b2976583ca83a","observation_id":"3bcf55f9-03af-4a07-bc39-bd4c46f401cb","resolution":{"observed_at":"2026-08-07T10:19:35.928792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00267","last_updated":"2024-09-03T14:01:54Z","snapshot_observed_at":"2026-07-06T16:13:07.384791Z","submitted_at":"2023-09-01T05:53:33Z","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00267","snapshot_observed_at":"2026-08-07T10:19:35.995780Z","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.995780Z"},"links":{"cited_paper":"/paper/2309.00267","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:84d2154876681183957e99bcf897b0f33405f46185aab73fd6b567d81cba9634","observation_id":"b18f9567-20f9-477a-9a02-cfd1064c7b01","resolution":{"observed_at":"2026-08-07T10:19:35.995780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-07T10:19:36.068399Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:36.068399Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:a72629a3335236a87e8311447f1019b142b9d33f4ca32aa3239967abe5d8c1c8","observation_id":"99d79685-81e9-4773-abbd-48ea36bbf58a","resolution":{"observed_at":"2026-08-07T10:19:36.068399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13787","last_updated":"2024-06-08T16:40:12Z","snapshot_observed_at":"2026-08-02T18:11:57.036767Z","submitted_at":"2024-03-20T17:49:54Z","title":"RewardBench: Evaluating Reward Models for Language Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13787","snapshot_observed_at":"2026-08-07T10:19:36.134562Z","title":"RewardBench: Evaluating Reward Models for Language Modeling,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:36.134562Z"},"links":{"cited_paper":"/paper/2403.13787","citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:be7c2459c6492aa83363b7a417e077476e23e400e8b11d4267809682638c873a","observation_id":"e97fd2f1-6d6e-44c7-9cc1-9fcf64545a28","resolution":{"observed_at":"2026-08-07T10:19:36.134562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.569163Z","title":"Iterative Reasoning Preference Optimization,","venue":null,"work_id":"894f2473-c89e-4329-853f-0ec027bc90b3","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:36.208057Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:6d3898ee6237d78fd45bf4c5cfcd035b706a01434161dc14b885b9b92d387008","observation_id":"3be384a4-a619-4978-9f80-881a09551d7f","resolution":{"observed_at":"2026-08-07T10:19:37.604055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:37.851865Z","title":"Available: https://neurips.cc/virtual/2024/108142","venue":null,"work_id":"db45c6e2-f833-49a3-8ffe-c54bcae052f2","year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.719817Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:d1a9c323a6822b47de1c7a5d41a4e6721146c0f2837bb14b746821d60298e165","observation_id":"16b69448-1360-4ae8-affb-5f829c56681e","resolution":{"observed_at":"2026-08-07T10:19:38.000462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:19:35.287341Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance","version":1},"reference_index":3836,"source":"pdf_text","source_observed_at":"2026-08-07T10:19:35.287341Z"},"links":{"citing_paper":"/paper/2506.05748"},"observation_digest":"sha256:297516423981d03a4ceeec6b5f542fb3210133e2d051d61625d6d2752e8cb59e","observation_id":"b797fe01-8b85-4336-b0b0-c115757d42f8","resolution":{"observed_at":"2026-08-07T10:19:35.287341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.05748","last_updated":"2025-06-06T05:18:54Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T21:55:18.527116Z","submitted_at":"2025-06-06T05:18:54Z","title":"Efficient Online RFT with Plug-and-Play LLM Judges: Unlocking State-of-the-Art Performance"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":20,"verified_exact":1,"verified_fuzzy":19},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2506.05748."}