{"as_of":"2026-08-21T08:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ef212583d8214d0a6d2bf0f5837259f0184afc7b768c21e042f77f32c9a7d3aa","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":44,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:33:19.906869Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-18T05:01:20.543826Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T23:33:10.824995Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2503.14476"},"observation_digest":"sha256:2e19d49307a2c8c798d324f435fd66b5d4e40d858d01705d002935a8e1d53e26","observation_id":"753f2d48-65e1-43e0-b6fa-b36884c6ae38","resolution":{"observed_at":"2026-05-22T23:35:13.471029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-08-14T16:01:52.772456Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T09:36:04.735688Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2504.05118"},"observation_digest":"sha256:250028e0fce525e5072bee70979d8a67457ff10ce20cf6975b987c5731ec2c0b","observation_id":"5c60ecf9-6bdd-416f-ad2c-0f8b063fc3fe","resolution":{"observed_at":"2026-05-13T09:36:04.792060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2504.12501","last_updated":"2026-08-03T01:47:58Z","snapshot_observed_at":"2026-08-16T12:27:55.152640Z","submitted_at":"2025-04-16T21:36:46Z","title":"Reinforcement Learning from Human Feedback","version":9},"reference_index":149,"source":"pdf_text","source_observed_at":"2026-05-22T19:27:40.991325Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2504.12501"},"observation_digest":"sha256:195b493e07ccc456f62a5c7a8c14742b0338580e407a47274bb97c9a04c52650","observation_id":"b87ec276-4947-4f84-ab2a-7014f9912e0b","resolution":{"observed_at":"2026-05-22T19:32:01.129490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-16T12:33:19.906869Z","title":"What’s behind PPO’s collapse in long-CoT? Value optimization holds the secret,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.12501","last_updated":"2026-08-03T01:47:58Z","snapshot_observed_at":"2026-08-16T12:27:55.152640Z","submitted_at":"2025-04-16T21:36:46Z","title":"Reinforcement Learning from Human Feedback","version":11},"reference_index":149,"source":"pdf_text","source_observed_at":"2026-08-16T12:33:19.906869Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2504.12501"},"observation_digest":"sha256:055a5ff64c0ab1b9e90223a6a897e1eafdab557a20378eb9979fe1f1f2b1f0f7","observation_id":"e4056bbd-5d2a-4514-81dd-aafef7c0700f","resolution":{"observed_at":"2026-08-16T12:33:19.906869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-08-15T17:20:54.840134Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T19:51:04.779597Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2504.20571"},"observation_digest":"sha256:a674906d5f570698531fc4e5e9d261011c7379d921f3a14c6de432571b600eb4","observation_id":"fac2e47b-4999-44bc-8193-6a25d9126a7b","resolution":{"observed_at":"2026-05-15T19:51:04.900537Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-16T05:12:18.680956Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.21277","last_updated":"2025-05-21T06:08:31Z","snapshot_observed_at":"2026-08-20T10:56:49.638012Z","submitted_at":"2025-04-30T03:14:28Z","title":"Reinforced MLLM: A Survey on RL-Based Reasoning in Multimodal Large Language Models","version":2},"reference_index":137,"source":"pdf_text","source_observed_at":"2026-08-16T05:12:18.680956Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2504.21277"},"observation_digest":"sha256:4aa81830ed1708ffe3d4d24d0d5cfa565d62e78022f70dc02f25d2d1cff65725","observation_id":"bc970383-16f6-4198-8620-b50c5f323077","resolution":{"observed_at":"2026-08-16T05:12:18.680956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-16T04:43:44.692497Z","title":"What's behind ppo's collapse in long-cot? value optimization holds the secret","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.00551","last_updated":"2025-05-15T14:16:03Z","snapshot_observed_at":"2026-08-17T18:29:23.516617Z","submitted_at":"2025-05-01T14:28:35Z","title":"100 Days After DeepSeek-R1: A Survey on Replication Studies and More Directions for Reasoning Language Models","version":3},"reference_index":153,"source":"arxiv_source","source_observed_at":"2026-08-16T04:43:44.692497Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2505.00551"},"observation_digest":"sha256:a9e0b4687de57fbde1ce945c2d9d8ed5df098084456f32ed7d12ef9d127f556f","observation_id":"8a248983-7160-41f9-bc5d-cbb54a625ba1","resolution":{"observed_at":"2026-08-16T04:43:44.692497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-20T20:37:36.775968Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"reference_index":168,"source":"pdf_text","source_observed_at":"2026-05-11T05:26:04.960844Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2505.07062"},"observation_digest":"sha256:21e21f98d98f4fd514d7cf464f99aa601f0d894483535f4e384efe7815aa6236","observation_id":"cdbef30b-f367-4bef-a478-724a06886ebe","resolution":{"observed_at":"2026-05-11T05:26:05.841675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-07T14:47:59.582127Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.17667","last_updated":"2025-05-27T09:39:47Z","snapshot_observed_at":"2026-08-18T11:46:25.087648Z","submitted_at":"2025-05-23T09:31:55Z","title":"QwenLong-L1: Towards Long-Context Large Reasoning Models with Reinforcement Learning","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:47:59.582127Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2505.17667"},"observation_digest":"sha256:17f2a15531dd5aac631aec0dad66ae303d63edbcfe183891e79a9994b0a3eb7d","observation_id":"30caf4b3-60a0-444d-8f18-cbce39e87a96","resolution":{"observed_at":"2026-08-07T14:47:59.582127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-07T14:31:12.361455Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18536","last_updated":"2025-05-24T06:01:48Z","snapshot_observed_at":"2026-08-16T23:13:31.947741Z","submitted_at":"2025-05-24T06:01:48Z","title":"Reinforcement Fine-Tuning Powers Reasoning Capability of Multimodal Large Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:31:12.361455Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2505.18536"},"observation_digest":"sha256:2b40ba8518ebc647fd8a916283fc794cbdf2c5613ab929209b265bc1a01bffff","observation_id":"8a89e05c-8fec-4938-b143-5e4d83dab744","resolution":{"observed_at":"2026-08-07T14:31:12.361455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-07T14:11:08.843880Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret.arXiv preprint arXiv:2503.01491, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19914","last_updated":"2025-06-09T07:49:32Z","snapshot_observed_at":"2026-08-07T16:09:17.796555Z","submitted_at":"2025-05-26T12:40:31Z","title":"Enigmata: Scaling Logical Reasoning in Large Language Models with Synthetic Verifiable Puzzles","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:08.843880Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2505.19914"},"observation_digest":"sha256:75fc5e572cfecb85b973bb1b6c41cfa3898903a72a0d0008058249b031d537af","observation_id":"7bac26cf-a7c9-457e-b4af-6384c8746ff6","resolution":{"observed_at":"2026-08-07T14:11:08.843880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-07T04:39:38.437982Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret.arXiv preprint arXiv:2503.01491, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10406","last_updated":"2025-06-12T06:59:35Z","snapshot_observed_at":"2026-08-15T05:52:57.120730Z","submitted_at":"2025-06-12T06:59:35Z","title":"PAG: Multi-Turn Reinforced LLM Self-Correction with Policy as Generative Verifier","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T04:39:38.437982Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2506.10406"},"observation_digest":"sha256:c73f8330c8dde72b1bf212d311e8c7b3135c8baf961e1c79c5359fc6d7e304e2","observation_id":"41185ef1-a4a3-4ab5-8244-c3389629e608","resolution":{"observed_at":"2026-08-07T04:39:38.437982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2507.01679","last_updated":"2026-05-15T06:56:26Z","snapshot_observed_at":"2026-08-15T15:20:51.338738Z","submitted_at":"2025-07-02T13:04:09Z","title":"Blending Supervised and Reinforcement Fine-Tuning with Prefix Sampling","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-21T23:39:39.018498Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2507.01679"},"observation_digest":"sha256:c5ac900451d1f9b7b28884cd760ee15135a27869f8246e49c7668473e1d3bc0a","observation_id":"0359e4e7-fb51-4003-a55b-23be9a60d1b9","resolution":{"observed_at":"2026-05-21T23:40:46.430743Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-06T15:55:42.604562Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.14683","last_updated":"2025-07-19T16:21:23Z","snapshot_observed_at":"2026-08-19T12:08:46.453087Z","submitted_at":"2025-07-19T16:21:23Z","title":"MiroMind-M1: An Open-Source Advancement in Mathematical Reasoning via Context-Aware Multi-Stage Policy Optimization","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:55:42.604562Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2507.14683"},"observation_digest":"sha256:d3d6c24d27ff3484e77c2e98a434ad0f248e69914611e49c4684acbd5ccc2efd","observation_id":"d3e2fe4e-2353-4b39-b05f-cf96aee9919c","resolution":{"observed_at":"2026-08-06T15:55:42.604562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2509.02544","last_updated":"2025-09-05T14:59:27Z","snapshot_observed_at":"2026-08-18T10:03:05.306613Z","submitted_at":"2025-09-02T17:44:45Z","title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-13T10:13:58.774968Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2509.02544"},"observation_digest":"sha256:cb182519b18a3be94001e820227ef977c2842b5c3d0f32e2f3e7af80d61309f5","observation_id":"65780489-24fc-4086-8b40-c06b37a4b90a","resolution":{"observed_at":"2026-05-13T10:13:58.979808Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T10:31:37.938067Z","title":"What's behind ppo's collapse in long-cot? value optimization holds the secret","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.04027","last_updated":"2026-06-09T07:28:31Z","snapshot_observed_at":"2026-08-18T16:28:53.247978Z","submitted_at":"2025-09-04T09:02:16Z","title":"Why Does Reasoning Length Converge? Unveiling the Underfitting-Overfitting Trade-off in Chain-of-Thought","version":4},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T10:31:37.938067Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2509.04027"},"observation_digest":"sha256:2f9c42f670f1c27c4ea0d8244ced08daf750aa62ed92520e70394a637b5d2fd6","observation_id":"5c9667f0-812e-4e67-8d41-0e8f7003c117","resolution":{"observed_at":"2026-08-05T10:31:37.938067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T05:45:04.288900Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05007","last_updated":"2025-09-08T03:26:03Z","snapshot_observed_at":"2026-08-16T11:58:24.387495Z","submitted_at":"2025-09-05T11:14:11Z","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T05:45:04.288900Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2509.05007"},"observation_digest":"sha256:274289a92681ea144822bdf5a8eab58b887b2f45a851e2a0e65ad25c45d1670a","observation_id":"9a039030-0780-4570-af85-a240679e1ace","resolution":{"observed_at":"2026-08-05T05:45:04.288900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2510.00568","last_updated":"2026-05-08T11:30:29Z","snapshot_observed_at":"2026-08-13T17:37:40.536432Z","submitted_at":"2025-10-01T06:44:28Z","title":"ReSeek: A Self-Correcting Framework for Search Agents with Instructive Rewards","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-18T11:20:02.118579Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2510.00568"},"observation_digest":"sha256:6bb59b0dab44fcea3f36801fc09fbf055b0d530b7c21e9216847d3f37ec193a5","observation_id":"0fa64f27-2935-4dae-ba5a-1f772792bbb3","resolution":{"observed_at":"2026-05-18T11:21:18.679910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2510.13786","last_updated":"2025-10-15T17:43:03Z","snapshot_observed_at":"2026-08-18T16:44:13.851821Z","submitted_at":"2025-10-15T17:43:03Z","title":"The Art of Scaling Reinforcement Learning Compute for LLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T16:29:13.954029Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2510.13786"},"observation_digest":"sha256:eaa251125db9b488ad162e7b59077a8fb7e8bc3466c93c5c6dcdf9f060a81303","observation_id":"ac9027f1-458d-4505-a4d9-19450ba8c1d2","resolution":{"observed_at":"2026-05-16T16:29:14.050525Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2511.07833","last_updated":"2026-05-09T19:03:03Z","snapshot_observed_at":"2026-08-20T09:26:23.689219Z","submitted_at":"2025-11-11T05:03:22Z","title":"MURPHY: Feedback-Aware GRPO with Retrospective Credit Assignment for Multi-Turn Code Generation","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-17T23:13:43.754235Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2511.07833"},"observation_digest":"sha256:4328125775074d92105a6c0a434679419e3153ba6551a8400dc45a0d1f435496","observation_id":"04809f1f-fea8-4b58-a2c3-8888db5b6d5a","resolution":{"observed_at":"2026-05-17T23:15:26.747000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-03T17:15:17.712961Z","title":"What’s behind ppo’s collapse in long-cot? value opti- mization holds the secret.arXiv preprint arXiv:2503.01491,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.10414","last_updated":"2026-05-30T03:42:16Z","snapshot_observed_at":"2026-08-15T18:36:58.856582Z","submitted_at":"2025-12-11T08:27:02Z","title":"Boosting RL-Based Visual Reasoning with Selective Adversarial Entropy Intervention","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-03T17:15:17.712961Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2512.10414"},"observation_digest":"sha256:41821ee23b4ddfded81ba471f9776bec1ecd6f8e57179f8bf47a9f9925950d1c","observation_id":"20b51108-69a6-4faf-9107-233b10865420","resolution":{"observed_at":"2026-08-03T17:15:17.712961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2602.01970","last_updated":"2026-05-15T12:23:06Z","snapshot_observed_at":"2026-08-18T19:27:56.462164Z","submitted_at":"2026-02-02T11:24:36Z","title":"Small Generalizable Prompt Predictive Models Can Steer Efficient RL Post-Training of Large Reasoning Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-21T14:09:26.842696Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2602.01970"},"observation_digest":"sha256:91911a53f37daf8b110a948cdc55ea7ecf0b66df9f78a1fbcece3e1b04207d0c","observation_id":"31ef4ebc-f0b0-47b0-985e-68fa4d6f064b","resolution":{"observed_at":"2026-05-21T14:10:13.030895Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-02T19:53:08.739032Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret.arXiv preprint arXiv:2503.01491,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.00963","last_updated":"2026-06-01T02:47:36Z","snapshot_observed_at":"2026-08-21T03:06:40.748876Z","submitted_at":"2026-03-01T07:40:12Z","title":"Stabilizing Policy Optimization via Logits Convexity","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T19:53:08.739032Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2603.00963"},"observation_digest":"sha256:9eb990318af4b1a05d8c0918dcfbc310cbfa3cd54bd1565a6b94507c0950d2e9","observation_id":"7951a39a-8e29-4105-9e88-6df39b8ef26c","resolution":{"observed_at":"2026-08-02T19:53:08.739032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2604.10701","last_updated":"2026-04-12T15:54:11Z","snapshot_observed_at":"2026-08-18T00:53:41.672740Z","submitted_at":"2026-04-12T15:54:11Z","title":"Bringing Value Models Back: Generative Critics for Value Modeling in LLM Reinforcement Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T15:09:29.657563Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2604.10701"},"observation_digest":"sha256:0188bd62ca5ee5f6b00336016e37f818460abea88e41cf998d4403eb91a920ce","observation_id":"66db204c-ed65-47a9-92ef-c982aa62aa79","resolution":{"observed_at":"2026-05-11T11:06:05.345214Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2604.13010","last_updated":"2026-05-08T06:38:25Z","snapshot_observed_at":"2026-08-16T21:35:51.236868Z","submitted_at":"2026-04-14T17:44:50Z","title":"Lightning OPD: Efficient Post-Training for Large Reasoning Models with Offline On-Policy Distillation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T15:12:06.985609Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2604.13010"},"observation_digest":"sha256:00a2fd4e8a925487501de05f960aa8a91c77ec7b6f0d02a4458972910dffff78","observation_id":"3bb9ead0-a0e3-4ebe-9f8d-7d3751944c49","resolution":{"observed_at":"2026-05-11T11:01:08.079209Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2604.13010","last_updated":"2026-05-08T06:38:25Z","snapshot_observed_at":"2026-08-16T21:35:51.236868Z","submitted_at":"2026-04-14T17:44:50Z","title":"Lightning OPD: Efficient Post-Training for Large Reasoning Models with Offline On-Policy Distillation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T01:04:12.454268Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2604.13010"},"observation_digest":"sha256:7347f048ffbe858a1b24253bf1001144c81564dbd9db5c0a777ee3b0e4a270b6","observation_id":"94558f63-35ae-47fd-8d26-779ff6493a39","resolution":{"observed_at":"2026-05-11T04:50:54.566752Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2604.13197","last_updated":"2026-08-04T16:37:12Z","snapshot_observed_at":"2026-08-17T01:52:32.626782Z","submitted_at":"2026-04-14T18:19:54Z","title":"Unleashing Implicit Rewards: Prefix-Value Learning for Distribution-Level Optimization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T15:45:15.962965Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2604.13197"},"observation_digest":"sha256:4336414ddae8b2066857fd1520d474b21788cb2cdfa13fbce075f76694e0a8fd","observation_id":"bc66f473-1bf1-4c82-93a1-2f496ac34c44","resolution":{"observed_at":"2026-05-11T09:56:01.737351Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-07-12T20:58:19.104482Z","title":"net/forum?id=2a36EMSSTp","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.13197","last_updated":"2026-08-04T16:37:12Z","snapshot_observed_at":"2026-08-17T01:52:32.626782Z","submitted_at":"2026-04-14T18:19:54Z","title":"Unleashing Implicit Rewards: Prefix-Value Learning for Distribution-Level Optimization","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T20:58:19.104482Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2604.13197"},"observation_digest":"sha256:2f819bd87695fb0624f6b5bcdf8d28b9649028a6bdd933d6d6f53b6eb72166ae","observation_id":"f2228a7e-ff07-46f9-9d1f-8490b81e1d2f","resolution":{"observed_at":"2026-07-12T20:58:19.104482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2605.01327","last_updated":"2026-05-07T08:09:44Z","snapshot_observed_at":"2026-08-13T01:17:32.478140Z","submitted_at":"2026-05-02T08:47:45Z","title":"Segment-Aligned Policy Optimization for Multi-Modal Reasoning","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T14:44:31.160543Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2605.01327"},"observation_digest":"sha256:4b21117fa749ce611ecf08567fe4be474f77787e3284a4fc9f37cdbe117735e1","observation_id":"2fe4635e-d5d1-4b5f-99aa-f632a0fb50c4","resolution":{"observed_at":"2026-05-11T16:51:07.895095Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2605.07244","last_updated":"2026-05-08T05:01:40Z","snapshot_observed_at":"2026-08-15T21:47:26.504858Z","submitted_at":"2026-05-08T05:01:40Z","title":"Experience Sharing in Mutual Reinforcement Learning for Heterogeneous Language Models","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-11T02:02:41.411795Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2605.07244"},"observation_digest":"sha256:6e00010a24441d763009ecbbf0067470f7c18298e9677ca7a602c9d9181c4309","observation_id":"e12b29cd-60a4-4113-aa62-cab3f6236696","resolution":{"observed_at":"2026-05-11T04:00:55.031255Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2605.08905","last_updated":"2026-05-09T11:57:25Z","snapshot_observed_at":"2026-08-11T11:04:38.944000Z","submitted_at":"2026-05-09T11:57:25Z","title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:33.143247Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2605.08905"},"observation_digest":"sha256:d73b237ad220b765c6a28f74fa42d7f55318487e87167b6b6c0093f952cfbe87","observation_id":"b3fcf1df-66e4-4b00-a07b-b03b4210df44","resolution":{"observed_at":"2026-05-12T02:46:18.933292Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2605.12058","last_updated":"2026-05-21T13:10:04Z","snapshot_observed_at":"2026-08-17T23:14:19.646825Z","submitted_at":"2026-05-12T12:45:03Z","title":"Holder Policy Optimisation","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-13T06:08:28.855671Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2605.12058"},"observation_digest":"sha256:67aa02b31b05c7a4f338e9a294ef6cc8940d7a44532331e23e5e25308f94c1f0","observation_id":"26dab740-55b7-4829-91d6-748d3a6f615c","resolution":{"observed_at":"2026-05-13T06:12:22.747759Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2605.12058","last_updated":"2026-05-21T13:10:04Z","snapshot_observed_at":"2026-08-17T23:14:19.646825Z","submitted_at":"2026-05-12T12:45:03Z","title":"Holder Policy Optimisation","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-22T10:00:58.600743Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2605.12058"},"observation_digest":"sha256:b0043e32f8a6d016dfb2325c73650979a3fbbd36f76a57697c09e2f9c18b5627","observation_id":"1394ac26-6e52-457a-a261-5452c7f99f99","resolution":{"observed_at":"2026-05-22T10:01:22.994765Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2605.25582","last_updated":"2026-06-04T14:40:56Z","snapshot_observed_at":"2026-08-14T03:13:59.727067Z","submitted_at":"2026-05-25T08:32:24Z","title":"Extreme Region Policy Distillation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T23:14:56.223606Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2605.25582"},"observation_digest":"sha256:e5b84d78e7ed648233d657e881d9efeb7c670673cc84fc59795bed3c89364ce5","observation_id":"b9cbef82-41a3-4267-abd8-159305b64d18","resolution":{"observed_at":"2026-06-29T23:24:02.202182Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-08-12T03:28:30.781632Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":207,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:d1a3b7816efc97b714b8cd33cc15b03c66d34b45b8ce6db3f47be06923cd7d5c","observation_id":"3e59a4e0-5516-4789-bd87-45ea1d5982c1","resolution":{"observed_at":"2026-07-01T20:56:13.277692Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2606.01281","last_updated":"2026-05-31T15:06:38Z","snapshot_observed_at":"2026-08-19T05:16:52.771821Z","submitted_at":"2026-05-31T15:06:38Z","title":"RLVR without Ineffective Samples: Group Prioritized Off-Policy Optimization for LLM Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-28T17:25:50.758630Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2606.01281"},"observation_digest":"sha256:20935c62b898ee127804c8f74d9cbaa7c3a63a31299f24ce58ec9bb57c96d59e","observation_id":"efc2402c-5688-441b-b8ca-debbd664132b","resolution":{"observed_at":"2026-07-01T21:16:13.368996Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2606.03102","last_updated":"2026-06-02T03:42:04Z","snapshot_observed_at":"2026-08-11T22:46:29.104838Z","submitted_at":"2026-06-02T03:42:04Z","title":"Small RL Controller, Large Language Model: RL-Guided Adaptive Sampling for Test-Time Scaling","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-06-28T10:25:10.559953Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2606.03102"},"observation_digest":"sha256:43ea58418b117f8634fd49d5040327bbc621cde5adedcf935bb524e980073fc3","observation_id":"bf00ac9e-5b34-4693-8441-de58a793eace","resolution":{"observed_at":"2026-07-02T02:56:30.041691Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2606.20008","last_updated":"2026-06-18T09:44:12Z","snapshot_observed_at":"2026-07-06T23:55:15.440382Z","submitted_at":"2026-06-18T09:44:12Z","title":"VIMPO: Value-Implicit Policy Optimization for LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T17:59:29.066879Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2606.20008"},"observation_digest":"sha256:533725b0a6de24ce45d6ce880d4b54fb87de50cf2cd8dcd81e732ad8b9e6c8da","observation_id":"04f6656b-bf9f-42e5-9081-5c27fb4c390a","resolution":{"observed_at":"2026-07-04T03:29:30.690053Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-08-15T21:55:25.625601Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":254,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:c4d493e7d077475c7e9135a8c10bfbfaa2b4c3fac551789a6ddadb9061b8a1b6","observation_id":"906b2229-b53a-49da-92d2-0d6d5f83132c","resolution":{"observed_at":"2026-07-04T08:09:40.705178Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2607.05378","last_updated":"2026-07-06T17:55:12Z","snapshot_observed_at":"2026-08-14T12:36:48.351389Z","submitted_at":"2026-07-06T17:55:12Z","title":"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-07T13:57:33.821970Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2607.05378"},"observation_digest":"sha256:1cb73776bb04328680d6daa3e32dc45471e3be27cff7f0404b161baa82e5be53","observation_id":"6bbf5e80-5a18-4823-a0c3-2e3b10152ac7","resolution":{"observed_at":"2026-07-07T14:03:48.691291Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":"2503.01491","doi":"10.48550/arxiv.2503.01491","metadata_source":"pith","pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What’s behind ppo’s collapse in long-cot? value optimization holds the secret","venue":"cs.LG","work_id":"b9f345a1-11ca-4214-b731-220afdbef297","year":2025},"citing_paper":{"arxiv_id":"2607.05378","last_updated":"2026-07-06T17:55:12Z","snapshot_observed_at":"2026-08-14T12:36:48.351389Z","submitted_at":"2026-07-06T17:55:12Z","title":"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-07T13:57:33.821970Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2607.05378"},"observation_digest":"sha256:14221f3b3d4a5c7e1c7c9eafcbc0c4c810cbb0676bfa6d87ba172b7c6b8d269f","observation_id":"b0504468-7755-4dd9-a45f-4f2ae2ccb475","resolution":{"observed_at":"2026-07-07T14:03:48.463323Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T16:19:05.612581+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-07-31T07:02:36.706606Z","title":"What’s behind PPO’s collapse in long-CoT? value optimization holds the secret.arXiv preprint arXiv:2503.01491,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28449","last_updated":"2026-07-30T16:17:15Z","snapshot_observed_at":"2026-08-13T10:47:03.640978Z","submitted_at":"2026-07-30T16:17:15Z","title":"Lightning OPD 2.0: Mitigating Style Bias in Cross-Teacher On-Policy Distillation for Large Reasoning Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-31T07:02:36.706606Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2607.28449"},"observation_digest":"sha256:1ef8b24591eea003278f4ff3a0c69a13b3777911e7f30731a760973f811b5b65","observation_id":"eef9deb5-9289-4a84-8567-56f20e255c95","resolution":{"observed_at":"2026-07-31T07:02:36.706606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-08T01:03:50.895805Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03068","last_updated":"2026-08-04T03:30:59Z","snapshot_observed_at":"2026-08-20T05:58:18.832350Z","submitted_at":"2026-08-04T03:30:59Z","title":"CVPO: Enhancing LLM Reinforcement Learning Reasoning via Value-Variance Adaptation and Dynamic Curriculum Learning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T01:03:50.895805Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2608.03068"},"observation_digest":"sha256:280cd3a049b2dcc3eadf1c026b23361e29b83f0cffc1252a40dca8c235f41445","observation_id":"47b2cca1-ef7f-429d-a4b9-b4f81fd44a93","resolution":{"observed_at":"2026-08-08T01:03:50.895805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01491","snapshot_observed_at":"2026-08-12T00:20:30.992560Z","title":"Yufeng Yuan, Yu Yue, Ruofei Zhu, Tiantian Fan, and Lin Yan","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08255","last_updated":"2026-08-08T17:32:34Z","snapshot_observed_at":"2026-08-18T07:47:13.007916Z","submitted_at":"2026-08-08T17:32:34Z","title":"Learning from Environmental Feedback: Credit Assignment across Multiple Timescales for Agentic Reinforcement Learning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T00:20:30.992560Z"},"links":{"cited_paper":"/paper/2503.01491","citing_paper":"/paper/2608.08255"},"observation_digest":"sha256:60e19907b4bf1bcfeca6414ac19320acb8379740ed0ee48181f8b17095b90761","observation_id":"af1ee5bb-ac15-47a5-a867-87075d555589","resolution":{"observed_at":"2026-08-12T00:20:30.992560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.01491/citation-record","integrity":"/paper/2503.01491/integrity","json":"/paper/2503.01491/citation-record.json","paper":"/paper/2503.01491"},"outbound":[],"paper":{"arxiv_id":"2503.01491","last_updated":"2025-03-03T12:59:25Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-19T09:54:01.929451Z","submitted_at":"2025-03-03T12:59:25Z","title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 44 inbound Pith citation observations for arXiv:2503.01491."}