{"as_of":"2026-08-07T22:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f14afef668c08d930f0854e494f23b00f49cdeab301908e3454dc67913bb762c","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:32:31.776781Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:07:30.092931Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T08:40:41.514103Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"cited_work":{"arxiv_id":"2506.02553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02553","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Response-level rewards are all you need for online reinforcement learning in llms: A mathematical perspective","venue":null,"work_id":"bd00da2d-e4a8-4dab-9a51-19d8603c4907","year":2025},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":252,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2506.02553","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:4dba68995edf6c73411d90b2537042cc42d2786613e7744247c8af5cdaeceb99","observation_id":"458b6a9e-4f12-408c-853b-786e6f2ba5ed","resolution":{"observed_at":"2026-05-12T08:40:41.517098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02553","snapshot_observed_at":"2026-08-04T16:07:30.092931Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.16679","last_updated":"2025-09-20T13:11:28Z","snapshot_observed_at":"2026-08-04T16:07:24.699834Z","submitted_at":"2025-09-20T13:11:28Z","title":"Reinforcement Learning Meets Large Language Models: A Survey of Advancements and Applications Across the LLM Lifecycle","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-04T16:07:30.092931Z"},"links":{"cited_paper":"/paper/2506.02553","citing_paper":"/paper/2509.16679"},"observation_digest":"sha256:cacb63b0353c3ea50dfcd326a91f16bd27a73e8d80b62ca1ec9bb4830aff27ec","observation_id":"cff38d40-0384-4dfd-b50f-4b7a1be48da5","resolution":{"observed_at":"2026-08-04T16:07:30.092931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"cited_work":{"arxiv_id":"2506.02553","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02553","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Response-level rewards are all you need for online reinforcement learning in llms: A mathematical perspective","venue":null,"work_id":"bd00da2d-e4a8-4dab-9a51-19d8603c4907","year":2025},"citing_paper":{"arxiv_id":"2605.10218","last_updated":"2026-05-11T08:58:40Z","snapshot_observed_at":"2026-07-06T23:22:13.774652Z","submitted_at":"2026-05-11T08:58:40Z","title":"Relative Score Policy Optimization for Diffusion Language Models","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-12T03:47:42.196931Z"},"links":{"cited_paper":"/paper/2506.02553","citing_paper":"/paper/2605.10218"},"observation_digest":"sha256:7f0481dbb466ff06198de7031eac3efacb2e555bfe2a973d1565bf953f2e62e5","observation_id":"389afa80-1000-430a-9628-73a046081c5c","resolution":{"observed_at":"2026-05-12T06:56:29.690474Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02553/citation-record","integrity":"/paper/2506.02553/integrity","json":"/paper/2506.02553/citation-record.json","paper":"/paper/2506.02553"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:32:28.119855Z","title":"Gpt-4o system card.arXiv preprint arXiv:2410.21276, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.119855Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:a91946c5bfd8baec17e1062cf515a5533c78b2cc3f5597a7fb36cb47ca87a9d8","observation_id":"a1610d62-8496-4528-b307-6534336b9c74","resolution":{"observed_at":"2026-08-07T11:32:28.119855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T11:32:28.178328Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.178328Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:09752fbb20a13576bc75b1f293328b3205127316cf0e5932cbb0aab1a6bf44a9","observation_id":"073987cd-41cf-4941-a263-ca7b0bcbdc3c","resolution":{"observed_at":"2026-08-07T11:32:28.178328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T11:32:28.272952Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.272952Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:1acfbbc1fbac1de884ff1c0e7adaaf0fd13932dab03f0cd3ba5b6591f309f0bd","observation_id":"d847781b-9233-4555-8232-c1e126abbad6","resolution":{"observed_at":"2026-08-07T11:32:28.272952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:33.247829Z","title":"The amazon nova family of models: Technical report and model card","venue":null,"work_id":"a2f3d7c8-274d-4c98-a64e-1c94d9a6fc72","year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.391741Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:b9cd82e704409c7c76c25aaefd344e56e0ff8cb0d10c3ba32edae161913e08ac","observation_id":"fac76498-5a66-438f-8048-394162622066","resolution":{"observed_at":"2026-08-07T11:32:33.438977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T11:32:28.487274Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.487274Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:1bd45a1ed8c53f47bbe2884c419632b776c8db12a8eda05b6c1dbfa991d1c8c8","observation_id":"8cfe9e74-aa5e-4339-a953-161932231c2a","resolution":{"observed_at":"2026-08-07T11:32:28.487274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T11:32:28.616012Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.616012Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:317c548f753b6c1ac2a0b1e200f5feda66f473c994cac42d753caf36c121f963","observation_id":"838fd653-fbef-4b82-b5c1-7cc2b920e801","resolution":{"observed_at":"2026-08-07T11:32:28.616012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T11:32:28.675759Z","title":"Proximal policy optimization algorithms.arXiv preprint arXiv:1707.06347, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.675759Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:97ea72a5913e643d1a9ae7040ab17b2019da603aa708430e3d781ee938de3f86","observation_id":"206e65fa-4c05-4e23-83ff-e6b70f85acf8","resolution":{"observed_at":"2026-08-07T11:32:28.675759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:28.759565Z","title":"Learning to summarize with human feedback.Advances in neural information processing systems, 33:3008–3021, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.759565Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:985c8ada4a48bfecbfe14836a5704bb944749f4c24551583ae068f7aaaa9e277","observation_id":"7a22d8ee-ed50-44dc-9392-7f7f9866d0bb","resolution":{"observed_at":"2026-08-07T11:32:28.759565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:28.849369Z","title":"Training language models to follow instructions with human feedback.Advances in neural information processing systems, 35:27730–27744, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.849369Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:dbef59b4226007c48cdb038823cb299fdbb2f4080c00f65d2d9ccfee31726060","observation_id":"db41d348-05c2-4c17-b586-f1c328306197","resolution":{"observed_at":"2026-08-07T11:32:28.849369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-07T11:32:28.936948Z","title":"Training a helpful and harmless assistant with reinforcement learning from human feedback.arXiv preprint arXiv:2204.05862, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:28.936948Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:71ad0a6a606d4ca0dfb53ad697a96cc32b92ef8d7226259f8e24c295afb7610b","observation_id":"fa568fdd-33ba-47a7-8f3a-c6a07f2bdd31","resolution":{"observed_at":"2026-08-07T11:32:28.936948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10505","last_updated":"2024-05-16T02:22:23Z","snapshot_observed_at":"2026-07-06T16:33:55.369704Z","submitted_at":"2023-10-16T15:25:14Z","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10505","snapshot_observed_at":"2026-08-07T11:32:29.011230Z","title":"Remax: A sim- ple, effective, and efficient reinforcement learning method for aligning large language models.arXiv preprint arXiv:2310.10505, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.011230Z"},"links":{"cited_paper":"/paper/2310.10505","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:5fae6af789e08fd7692a2f19cf5f3f98b87aff4feb08961e373c4110dade651d","observation_id":"b3b842df-d41e-4957-991c-f0ea6963c5f2","resolution":{"observed_at":"2026-08-07T11:32:29.011230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T11:32:29.157791Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.157791Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:e5b85680465e27f0e287019e427fd74f57a2e3bd710307e9ceaa988dfaf08483","observation_id":"92d7bb58-110a-4dc1-b1a8-d6fdf26249d2","resolution":{"observed_at":"2026-08-07T11:32:29.157791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-08-07T11:32:29.285863Z","title":"Back to basics: Revisiting reinforce style optimization for learning from human feedback in llms.arXiv preprint arXiv:2402.14740, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.285863Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:b3067966a98790c0ca71cecf6c73f4d46dd3a63d1b33b36205954b7fc4c284e3","observation_id":"ac4235ba-1686-4383-89d7-670ef940fb6a","resolution":{"observed_at":"2026-08-07T11:32:29.285863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18922","last_updated":"2025-05-21T15:34:02Z","snapshot_observed_at":"2026-07-31T18:34:20.947967Z","submitted_at":"2024-04-29T17:58:30Z","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18922","snapshot_observed_at":"2026-08-07T11:32:29.379167Z","title":"Dpo meets ppo: Reinforced token optimization for rlhf.arXiv preprint arXiv:2404.18922, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.379167Z"},"links":{"cited_paper":"/paper/2404.18922","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:613da14e69e611420ca478380fdbedd05e1e9fbfbcca136db06c322ebba17d02","observation_id":"c31fd329-18ae-4c4a-af35-5ef3063ac485","resolution":{"observed_at":"2026-08-07T11:32:29.379167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00782","last_updated":"2024-02-01T17:10:35Z","snapshot_observed_at":"2026-07-06T17:23:51.547578Z","submitted_at":"2024-02-01T17:10:35Z","title":"Dense Reward for Free in Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00782","snapshot_observed_at":"2026-08-07T11:32:29.474618Z","title":"Dense reward for free in reinforcement learning from human feedback.arXiv preprint arXiv:2402.00782, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.474618Z"},"links":{"cited_paper":"/paper/2402.00782","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:18a3a6c57084ea9f88e5518d9ebbb74738b69276206d7e6a5d7e4ed575f03e63","observation_id":"3d4739b4-8fbb-4f16-9d85-ca8afd503afd","resolution":{"observed_at":"2026-08-07T11:32:29.474618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01456","snapshot_observed_at":"2026-08-07T11:32:29.517440Z","title":"Process reinforcement through implicit rewards.arXiv preprint arXiv:2502.01456, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.517440Z"},"links":{"cited_paper":"/paper/2502.01456","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:72e9cbf923aa02b4e06cf0bce3cae7aef1aebd7dfa134adca3b72ed22380420e","observation_id":"9dc7d6f1-aa41-45d1-a570-87e7f284de1d","resolution":{"observed_at":"2026-08-07T11:32:29.517440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.08302","last_updated":"2025-09-11T10:17:06Z","snapshot_observed_at":"2026-08-05T19:19:51.785740Z","submitted_at":"2024-11-13T02:45:21Z","title":"RED: Unleashing Token-Level Rewards from Holistic Feedback via Reward Redistribution","version":2},"cited_work":{"arxiv_id":"2411.08302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.08302","snapshot_observed_at":"2026-08-07T11:32:32.239050Z","title":"RED: Unleashing Token-Level Rewards from Holistic Feedback via Reward Redistribution","venue":"cs.CL","work_id":"b790476c-97f6-4051-8e20-05fa884acb61","year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.571593Z"},"links":{"cited_paper":"/paper/2411.08302","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:5d6310d791db167a7ae034d2c610525bd27808d42a5d9f50649e2129d2e758b2","observation_id":"d1e21099-45e5-456f-9a22-a89f33de763d","resolution":{"observed_at":"2026-08-07T11:32:32.289654Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10719","last_updated":"2024-10-10T08:30:17Z","snapshot_observed_at":"2026-07-06T18:01:05.698046Z","submitted_at":"2024-04-16T16:51:53Z","title":"Is DPO Superior to PPO for LLM Alignment? A Comprehensive Study","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.10719","snapshot_observed_at":"2026-08-07T11:32:29.652497Z","title":"Is dpo superior to ppo for llm alignment? a comprehensive study.arXiv preprint arXiv:2404.10719, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.652497Z"},"links":{"cited_paper":"/paper/2404.10719","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:5068559f98dd88756235c5b77ee40e95f3483c0ae19f728807921eedbaa51475","observation_id":"226e8b7d-71ea-4786-aa9b-ea8cb8881666","resolution":{"observed_at":"2026-08-07T11:32:29.652497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:33.051822Z","title":"Reinforcement learning: An introduction","venue":null,"work_id":"3e8eb64a-b069-4cc4-a72e-787b7e3bf1a0","year":2018},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.736929Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:d6d6512b297cb654adcad6e6c2339cb475d8b853afe5d1fbfeb771278a4fb5a1","observation_id":"0f06674a-aab8-4c2f-ace5-3d413506452c","resolution":{"observed_at":"2026-08-07T11:32:33.106505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.22244","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.108585Z","title":"Analysis of on-policy policy gradient methods under the distribution mismatch.arXiv preprint arXiv:2503.22244, 2025","venue":null,"work_id":"4e6b86a9-119a-483d-bbfc-8eade381a577","year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.838106Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:0882900e903d4607965d9a75954eed6ff8da6645ac4298f6cbbda33fdefd51ee","observation_id":"e3eae744-e475-416a-b86c-3b11a18d485e","resolution":{"observed_at":"2026-08-07T11:32:32.175493Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.957520Z","title":"High-dimensional continu- ous control using generalized advantage estimation, 2018","venue":null,"work_id":"e523ba1c-ae48-4339-9d84-97ce562cfb00","year":2018},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:29.935825Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:974aea2df63e7cb4fcdb98a237e64c39091a6f912baee2222f3072a3591f73dc","observation_id":"ebc8704f-7ad7-42cb-b105-420a41cba4df","resolution":{"observed_at":"2026-08-07T11:32:32.986611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-08-05T20:06:13.107818Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-08-07T11:32:30.026404Z","title":"Vineppo: Unlocking rl potential for llm reasoning through refined credit assignment.arXiv preprint arXiv:2410.01679, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.026404Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:2dfa47e503a5bdb385cb6d46dcea5ae78a3df66922d719c60a6bcca53e16cb29","observation_id":"229e0672-ac94-41c0-aad3-7d737231e735","resolution":{"observed_at":"2026-08-07T11:32:30.026404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:30.149593Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.149593Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:784c8b682e1429805c0269f0875fb36cf8b291a9778fbd85bb1e0338851c9361","observation_id":"faea92d1-0529-408c-8cf3-62950293087a","resolution":{"observed_at":"2026-08-07T11:32:30.149593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.819233Z","title":"Direct preference optimization: Your language model is secretly a reward model.Advances in Neural Information Processing Systems, 36:53728–53741, 2023","venue":null,"work_id":"e4c6c81e-3f88-4bdd-b339-7a6e8bb1a6ca","year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.236850Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:1b1a6fda8ad4285bd882d117601c5799ea4757aad3198a89db2a862db9437658","observation_id":"8d855c2e-f274-4eab-97d1-d23f6041ab13","resolution":{"observed_at":"2026-08-07T11:32:32.869318Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15478","last_updated":"2025-03-19T17:55:08Z","snapshot_observed_at":"2026-08-07T16:51:01.687483Z","submitted_at":"2025-03-19T17:55:08Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15478","snapshot_observed_at":"2026-08-07T11:32:30.314881Z","title":"Sweet-rl: Training multi-turn llm agents on collaborative reasoning tasks.arXiv preprint arXiv:2503.15478, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.314881Z"},"links":{"cited_paper":"/paper/2503.15478","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:04e71b9ca2b0188e6de3663d346c771c8ed75d3446001915a3b0bd5c6d5d34ab","observation_id":"3d37a5d8-7487-4031-94a2-c63c0b08b245","resolution":{"observed_at":"2026-08-07T11:32:30.314881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14655","last_updated":"2024-12-02T12:37:46Z","snapshot_observed_at":"2026-07-06T18:18:43.328203Z","submitted_at":"2024-05-23T14:53:54Z","title":"Multi-turn Reinforcement Learning from Preference Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14655","snapshot_observed_at":"2026-08-07T11:32:30.406644Z","title":"Multi-turn reinforcement learning from preference human feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.406644Z"},"links":{"cited_paper":"/paper/2405.14655","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:7d2e043be8bf76e92f03f78ae56bbcb7c3d9c7e4163c4ea20b0594b76999a2d3","observation_id":"06c03cc1-5efd-4109-ad8b-30f0011fe3dc","resolution":{"observed_at":"2026-08-07T11:32:30.406644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18232","last_updated":"2023-11-30T03:59:31Z","snapshot_observed_at":"2026-08-07T02:51:02.413859Z","submitted_at":"2023-11-30T03:59:31Z","title":"LMRL Gym: Benchmarks for Multi-Turn Reinforcement Learning with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18232","snapshot_observed_at":"2026-08-07T11:32:30.490860Z","title":"Lmrl gym: Benchmarks for multi-turn reinforcement learning with language models.arXiv preprint arXiv:2311.18232, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.490860Z"},"links":{"cited_paper":"/paper/2311.18232","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:d32f231fcf4d7dbc2a96de474bafd739f8357405ee2ef168d5b581d6b199c047","observation_id":"4ef9d1d0-0259-4636-8775-80a7ff65f737","resolution":{"observed_at":"2026-08-07T11:32:30.490860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19446","last_updated":"2024-02-29T18:45:56Z","snapshot_observed_at":"2026-08-07T22:12:46.006835Z","submitted_at":"2024-02-29T18:45:56Z","title":"ArCHer: Training Language Model Agents via Hierarchical Multi-Turn RL","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19446","snapshot_observed_at":"2026-08-07T11:32:30.549942Z","title":"Archer: Training language model agents via hierarchical multi-turn rl.arXiv preprint arXiv:2402.19446, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.549942Z"},"links":{"cited_paper":"/paper/2402.19446","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:4440dbd6988870c22f3b0f0d09a70c0930773c502b4bc0c9f4f1bfe035c3ab68","observation_id":"f04a9702-4b69-44c0-ac3b-86915a75c762","resolution":{"observed_at":"2026-08-07T11:32:30.549942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16145","last_updated":"2024-12-25T18:54:02Z","snapshot_observed_at":"2026-07-06T20:11:06.248415Z","submitted_at":"2024-12-20T18:49:45Z","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16145","snapshot_observed_at":"2026-08-07T11:32:30.615635Z","title":"Offline reinforcement learning for llm multi-step reasoning.arXiv preprint arXiv:2412.16145, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.615635Z"},"links":{"cited_paper":"/paper/2412.16145","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:0cca8c04f15176c864e19eba635cac36a4e16982d228caac163159569b4574ac","observation_id":"40d8a2c3-061f-4d2d-ab59-133b144a810e","resolution":{"observed_at":"2026-08-07T11:32:30.615635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.11221","last_updated":"2025-06-23T05:32:12Z","snapshot_observed_at":"2026-08-07T18:14:28.926787Z","submitted_at":"2025-02-16T17:54:57Z","title":"PlanGenLLMs: A Modern Survey of LLM Planning Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.11221","snapshot_observed_at":"2026-08-07T11:32:30.682383Z","title":"Plangenllms: A modern survey of llm planning capabilities.arXiv preprint arXiv:2502.11221, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.682383Z"},"links":{"cited_paper":"/paper/2502.11221","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:2614a3aaf3f476434624bdb66dbd93032421e7560ebe921b387a84dd1f7d9813","observation_id":"46c7a99b-9411-4c48-9a88-2d440dd9bf5d","resolution":{"observed_at":"2026-08-07T11:32:30.682383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14762","last_updated":"2024-11-05T16:40:21Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T18:21:59Z","title":"MT-Bench-101: A Fine-Grained Benchmark for Evaluating Large Language Models in Multi-Turn Dialogues","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14762","snapshot_observed_at":"2026-08-07T11:32:30.761569Z","title":"Mt-bench-101: A fine-grained benchmark for evaluating large language models in multi-turn dialogues.arXiv preprint arXiv:2402.14762, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.761569Z"},"links":{"cited_paper":"/paper/2402.14762","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:6d6f57c9f9ee72d47faf99d5293faf3c9e897cf8516b12de68cae0d91af47b13","observation_id":"d316c10c-5751-4cb2-a82f-c5351ec05453","resolution":{"observed_at":"2026-08-07T11:32:30.761569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.658344Z","title":"Interactive evaluation for medical LLMs via task- oriented dialogue system","venue":null,"work_id":"9ee545f8-f127-4b66-b3a6-79977eaa464c","year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.841930Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:26f4752438b4fdd2f758c032d09b17817042b0b76c0a7504b47c7437d3a7a85c","observation_id":"4c96c6b6-cb23-4153-9448-1ad05c0b0ae9","resolution":{"observed_at":"2026-08-07T11:32:32.721910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:30.927433Z","title":"Let’s verify step by step, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:30.927433Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:2f27dc29eb1e9df5ed54e8d6b5c078f9e2a4f2241c234db19f03fec5c4d838e8","observation_id":"7ed50a14-fea9-4537-9fe4-15c531531cc2","resolution":{"observed_at":"2026-08-07T11:32:30.927433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:32.480645Z","title":"Unraveling rlhf and its variants: Progress and practical engineering insights","venue":null,"work_id":"80c0526d-a3f2-4b8b-a77e-b439ed2f7bb0","year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.056825Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:556f367317deff3964e143adc6447643ac776db54d7e89724a59732270de64eb","observation_id":"7a0a2145-8133-4678-ac20-be71fc5c50ae","resolution":{"observed_at":"2026-08-07T11:32:32.554729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-07T11:32:31.151918Z","title":"Solving math word problems with process-and outcome-based feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.151918Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:58b32c207ce5c2743e6412b4c44d7e5e3ffec95828b777ad61177802bd1e78d0","observation_id":"c4d243be-85b3-4deb-8dd7-850e0b14cfb7","resolution":{"observed_at":"2026-08-07T11:32:31.151918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:31.223415Z","title":"Fine-grained human feedback gives better rewards for language model training.Advances in Neural Information Processing Systems, 36:59008–59033, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.223415Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:f2d91a2db5c4da9bd76fdd3f9a4f850c694f5a7a81d2a682c7337c836db98485","observation_id":"ce13e47a-3bf7-42d3-bbf8-efa2c0227fb7","resolution":{"observed_at":"2026-08-07T11:32:31.223415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07382","last_updated":"2024-02-19T18:19:20Z","snapshot_observed_at":"2026-07-06T17:15:22.551637Z","submitted_at":"2024-01-14T22:05:11Z","title":"Beyond Sparse Rewards: Enhancing Reinforcement Learning with Language Model Critique in Text Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07382","snapshot_observed_at":"2026-08-07T11:32:31.310922Z","title":"Beyond sparse re- wards: Enhancing reinforcement learning with language model critique in text generation.arXiv preprint arXiv:2401.07382, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.310922Z"},"links":{"cited_paper":"/paper/2401.07382","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:883347bc7b770eb68219c0624544fc6142223435a98fe971280f1dda6d295046","observation_id":"86e6e972-8ff4-4c20-a920-2dfcddb4ecf9","resolution":{"observed_at":"2026-08-07T11:32:31.310922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16574","last_updated":"2024-12-08T14:22:13Z","snapshot_observed_at":"2026-07-06T18:50:43.893885Z","submitted_at":"2024-07-23T15:27:37Z","title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.16574","snapshot_observed_at":"2026-08-07T11:32:31.365581Z","title":"Tlcr: Token-level continuous reward for fine-grained reinforcement learning from human feedback.arXiv preprint arXiv:2407.16574, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.365581Z"},"links":{"cited_paper":"/paper/2407.16574","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:774c5e704fb34c222c23996da69e40c32d0e4c23994c0647a27656824b6419a7","observation_id":"70da07b1-0cf1-44f1-ba5c-0c3613c1b6ab","resolution":{"observed_at":"2026-08-07T11:32:31.365581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08935","last_updated":"2024-02-19T14:07:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-14T13:41:54Z","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08935","snapshot_observed_at":"2026-08-07T11:32:31.431529Z","title":"Math- shepherd: Verify and reinforce llms step-by-step without human annotations.arXiv preprint arXiv:2312.08935, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.431529Z"},"links":{"cited_paper":"/paper/2312.08935","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:ab814849da9f2c783a7345444d7a242e81006e9a87743e99681e134249d686ab","observation_id":"005c73bf-7ead-4a34-a89f-163671031cb4","resolution":{"observed_at":"2026-08-07T11:32:31.431529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-08-07T11:32:31.527587Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning.arXiv preprint arXiv:2502.06781, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.527587Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:c7524d1e29183e7e18039c0bb7b1baf7cac044593bc6dc7cc90044310959911e","observation_id":"867e3e6f-9b30-47c1-b77b-cd076a11bcef","resolution":{"observed_at":"2026-08-07T11:32:31.527587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:32:31.601560Z","title":"Chatbot arena: An open platform for evaluating llms by human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.601560Z"},"links":{"citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:7228397fa015e445594002670c0f4524c76ba1a6242e2a983fb8cea3a423e1f6","observation_id":"eb286a80-c534-4ce2-98f5-35563562af88","resolution":{"observed_at":"2026-08-07T11:32:31.601560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13006","last_updated":"2025-03-30T17:59:47Z","snapshot_observed_at":"2026-07-06T19:05:05.530925Z","submitted_at":"2024-08-23T11:49:01Z","title":"Systematic Evaluation of LLM-as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13006","snapshot_observed_at":"2026-08-07T11:32:31.693251Z","title":"Systematic evaluation of llm-as-a-judge in llm alignment tasks: Explainable metrics and diverse prompt templates.arXiv preprint arXiv:2408.13006, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.693251Z"},"links":{"cited_paper":"/paper/2408.13006","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:1cdf0544b41ffd7c76a04e7ae766102af14109f7b6e25ce168a22fa6ca030052","observation_id":"b472d78c-92ea-4f16-8e06-1c1a56c278ee","resolution":{"observed_at":"2026-08-07T11:32:31.693251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02743","last_updated":"2025-02-14T23:02:03Z","snapshot_observed_at":"2026-07-06T19:27:11.995338Z","submitted_at":"2024-10-03T17:55:13Z","title":"MA-RLHF: Reinforcement Learning from Human Feedback with Macro Actions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02743","snapshot_observed_at":"2026-08-07T11:32:31.776781Z","title":"Ma-rlhf: Reinforcement learning from human feedback with macro actions.arXiv preprint arXiv:2410.02743, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:32:31.776781Z"},"links":{"cited_paper":"/paper/2410.02743","citing_paper":"/paper/2506.02553"},"observation_digest":"sha256:b8cb60f0db44697507ee66eb036c95aa28786296435efe017dc007adbb66d19a","observation_id":"71bc59d4-7290-4a87-b902-305f8be53e54","resolution":{"observed_at":"2026-08-07T11:32:31.776781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02553","last_updated":"2025-06-03T07:44:31Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T11:19:28.410568Z","submitted_at":"2025-06-03T07:44:31Z","title":"Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":35,"verified_exact":1,"verified_fuzzy":6},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 3 inbound Pith citation observations for arXiv:2506.02553."}