{"as_of":"2026-08-06T07:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0dd787f1c805502005d79458f30a64b7f0632b876ddd7dc669e3265fcc202758","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T10:24:53.245739Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.22570/citation-record","integrity":"/paper/2606.22570/integrity","json":"/paper/2606.22570/citation-record.json","paper":"/paper/2606.22570"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"Scaling Learning Algorithms Towards","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:cc78f5c5196b7c8b88940c3fd1113f32674be1432f417ef5d823a77b379e2d78","observation_id":"65b611c5-969b-4456-a04e-a1a4daa120da","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"and Osindero, Simon and Teh, Yee Whye , journal =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:d1f706b7662bd79d9c6b8697673b8981ebe731924985cae4b33293580c4ec45f","observation_id":"e281e20f-6585-427b-90fb-5fba933440e2","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2016 , publisher=","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:c13279c363c9265b3452812c7650822271670d773281ecc2ca1bf379a6307d0d","observation_id":"6dab33e7-8946-4901-9c35-4c4fe40c962a","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:55582d4dbdb3d8b37c32feed07f511d5d4841a0dfa3d0533c91d92522f109af7","observation_id":"5ddd0cda-a0b2-4228-af60-eb7cfa26cef5","resolution":{"observed_at":"2026-07-04T09:09:43.606398Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:df45696e434eab7abaf36050e16b16ef45da88b5442ffbb40931f0f0f9a542d3","observation_id":"abfbcc2c-8d86-446d-ad8f-3243bd3a4b89","resolution":{"observed_at":"2026-07-04T09:09:43.601370Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:e6e6c44c99ebde5fdbeb671465fc1fe249be6ae1cbf0ba4928d385449f5eb163","observation_id":"d5688a91-f619-4749-a915-c9df7ffba541","resolution":{"observed_at":"2026-07-04T09:09:43.598925Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2504.05118","doi":"10.1109/access.2024.3384487","metadata_source":"pith","pith_arxiv_id":"2504.05118","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","venue":"cs.AI","work_id":"c2351652-65f7-47cd-ae80-dbcd72a6eb20","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2504.05118","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:623b76a120821c35555bbd29b535d842c810be4746c4509ee57cd972406641dd","observation_id":"c5527d84-9d29-4b75-9797-8f0b4cf0a6d0","resolution":{"observed_at":"2026-07-04T09:09:43.603828Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:1b723dbcb8868db9c4905c1d8782535937d2a7d11c506763ec0a344b7a9d1cba","observation_id":"f64637eb-4856-4e27-9f83-a6a2bc0cf1e7","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:b7a2b934e8948de29b9e37897bcb8c2570bc4a0ea972b4f9ea3463da718ad8f7","observation_id":"94281c76-f5a3-49bb-b08d-d9e94d32efe9","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2022.acl-long.78","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.18653/v1/2022.acl-long.78","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","work_id":"49e3838d-3ea2-4bd0-a106-aa1e6391e5b0","year":2022},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:1bb2622e5830d7e5b8911a5b8dc28e72ba81b91787904bd74df887ce56bcad81","observation_id":"67eada85-28c2-4a11-9974-de08626c402f","resolution":{"observed_at":"2026-06-26T10:29:18.566303Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-25T21:53:20.037434+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T21:53:20.037434+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2025 , note =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:aef2ac9f2d42da6384e4835b45186b688d6e80e5e641ee43b5f6b9650ad29196","observation_id":"7d6175e9-6795-4dc5-b415-5933968ce487","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2024 , journal =","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:7d5f63e67d06e544a02f4db4c72487963da42d77b7f8454f0d17c07bf0461224","observation_id":"3876e9d5-9ae3-4dfe-a9ef-e42e59fe95e4","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2021 , eprint=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:a586a54a8efdd7c79887be4c1a777d3aee622b6e9a90f7dbd75394ec008e5f6b","observation_id":"e8e3d195-0951-4d2f-a6f7-f69c6920f5c1","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:d9180b5e277332e498e0b3825fce0bfaef5d240cdd268b2eae2f97e6c5f9f3d6","observation_id":"ffc53ab1-a6cf-4127-8cb4-cfb9803b5d22","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-07-06T19:55:37.400185Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":"2411.15124","doi":"10.48550/arxiv.2411.15124","metadata_source":"pith","pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","venue":"cs.CL","work_id":"28c9dbea-056a-48c2-8000-85f809827e45","year":2024},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:1270810c9171e3656a0a8d14700c059b9f096a54cb11c9a794db31b0b694e7c1","observation_id":"5baf5c3b-b621-45ea-b8c5-cad94d3c6868","resolution":{"observed_at":"2026-07-04T09:09:43.596474Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":"2412.16720","doi":"10.48550/arxiv.2412.16720","metadata_source":"pith","pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"OpenAI o1 System Card","venue":"cs.AI","work_id":"68d3c334-0fc9-49e3-b7b0-a69afae933e2","year":2024},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:88a8a114283c2bc4e74ac35c24cc34abff92d8a6e1e3bef556235c219169ba86","observation_id":"2a207b72-137d-44d8-a353-f43b8e1a264a","resolution":{"observed_at":"2026-07-04T09:09:43.594060Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:b2691e5c623e75ac1997848fd6263e0896872c20015b460c79bdb56386a35247","observation_id":"78689bf4-fd35-4f36-9e1e-660ffcdbe1b3","resolution":{"observed_at":"2026-07-04T09:09:43.589102Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"2025 , month =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:fba95141f6852a61abaa2c9ba5d29f4b6b01c7d40eb0a6efb22e6af613401ec8","observation_id":"ccb7237e-c26c-4967-87d8-688763bea325","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-07-06T21:28:24.644692Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.16400","doi":"10.48550/arxiv.2505.16400","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16400","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Acereason-nemotron: Advancing math and code reasoning through reinforcement learning","venue":"ArXiv.org","work_id":"428ad314-c120-41da-9db7-b8bc1918fffb","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2505.16400","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:9b9764404769069a974e6b063ce297199af8e6f868168f271f8f4b5cf286f459","observation_id":"4afb7a0a-b48a-4c0a-ab97-a7d8e16d28ed","resolution":{"observed_at":"2026-07-04T09:09:43.581253Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:a30a33e4ae7141094d006fd48b66b34f5d2167b6f7536d35a13151826c63a388","observation_id":"1b693f92-b6ab-4bdb-b1a9-9c23ed2fd224","resolution":{"observed_at":"2026-07-04T09:09:43.605643Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.15115","doi":"10.1145/3581783.3612503","metadata_source":"pith","pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5 Technical Report","venue":"cs.CL","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","year":2024},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:4155b8a7b85228fecb04b0350fc5279fe1d943e0bc100015f3bf7e212272fad5","observation_id":"2ef7407c-c595-4162-b127-8a0e10eadcf3","resolution":{"observed_at":"2026-07-04T09:09:43.586643Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01939","last_updated":"2025-11-13T10:08:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T17:54:39Z","title":"Beyond the 80/20 Rule: High-Entropy Minority Tokens Drive Effective Reinforcement Learning for LLM Reasoning","version":2},"cited_work":{"arxiv_id":"2506.01939","doi":"10.48550/arxiv.2506.01939","metadata_source":"pith","pith_arxiv_id":"2506.01939","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond the 80/20 Rule: High-Entropy Minority Tokens Drive Effective Reinforcement Learning for LLM Reasoning","venue":"cs.CL","work_id":"e5e936f3-0cff-4732-b394-f607d7a63f5f","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2506.01939","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:6db57613e01bd690d574db589fe606b8e107317ad3872c5f7af09c1ab9f9d67c","observation_id":"bde99d6b-5f49-4a85-a388-c870a2042597","resolution":{"observed_at":"2026-07-04T09:09:43.591750Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:08.024018+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:08.024018+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.12929","last_updated":"2025-05-19T10:14:08Z","snapshot_observed_at":"2026-07-06T21:26:11.353127Z","submitted_at":"2025-05-19T10:14:08Z","title":"Do Not Let Low-Probability Tokens Over-Dominate in RL for LLMs","version":1},"cited_work":{"arxiv_id":"2505.12929","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.12929","snapshot_observed_at":"2026-07-10T14:27:07.373132Z","title":"arXiv preprint arXiv:2505.12929 , year=","venue":"cs.CL","work_id":"965c4b96-d845-4417-bb01-9ec7fd03e93b","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2505.12929","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:ee2dfd5fa3a1887ba529d1c6618a1a39e48b14b71f9f458d95db81d5d9599477","observation_id":"af7bf6e4-7168-4bd3-8cb3-3bc0f4d082bd","resolution":{"observed_at":"2026-07-04T09:09:43.578480Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.14758","last_updated":"2025-11-08T04:52:16Z","snapshot_observed_at":"2026-07-06T21:43:45.343005Z","submitted_at":"2025-06-17T17:54:03Z","title":"Reasoning with Exploration: An Entropy Perspective","version":4},"cited_work":{"arxiv_id":"2506.14758","doi":"10.48550/arxiv.2506.14758","metadata_source":"pith","pith_arxiv_id":"2506.14758","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reasoning with Exploration: An Entropy Perspective","venue":"cs.CL","work_id":"5670d60c-ec64-40eb-93e1-1e51c35e9050","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2506.14758","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:220d359884cb80c7002b3cf70ca451cb2050fc16224826f342af5010799bc107","observation_id":"53c442e5-1ab9-45a7-a1ac-a0118e30868b","resolution":{"observed_at":"2026-07-04T09:09:43.563501Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22617","last_updated":"2025-05-28T17:38:45Z","snapshot_observed_at":"2026-07-06T21:32:23.537939Z","submitted_at":"2025-05-28T17:38:45Z","title":"The Entropy Mechanism of Reinforcement Learning for Reasoning Language Models","version":1},"cited_work":{"arxiv_id":"2505.22617","doi":"10.48550/arxiv.2505.22617","metadata_source":"pith","pith_arxiv_id":"2505.22617","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Entropy Mechanism of Reinforcement Learning for Reasoning Language Models","venue":"cs.LG","work_id":"d4b4aee4-d20f-4572-886a-4ba9ea6c9b81","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2505.22617","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:56686fead0901845aa06f358b43aa335f837bc97d6b6df7836a1dc8456a4a11c","observation_id":"91c6b4c4-7983-4953-b57b-0c2be8bfef38","resolution":{"observed_at":"2026-07-04T09:09:43.569466Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20282","last_updated":"2025-08-21T08:51:49Z","snapshot_observed_at":"2026-07-06T21:30:52.246162Z","submitted_at":"2025-05-26T17:58:30Z","title":"One-shot Entropy Minimization","version":4},"cited_work":{"arxiv_id":"2505.20282","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.20282","snapshot_observed_at":"2026-07-04T09:09:43.563954Z","title":"One-shot entropy minimization.arXiv preprint arXiv:2505.20282","venue":null,"work_id":"5336276b-09e9-4d0a-8cce-c6e5dde7e628","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2505.20282","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:76a3767f0432faaae1c36fd16671bff38ec528e1e07d202c6cfbf7fc58858e5f","observation_id":"2b142ae2-05d1-4a9e-bfb1-53ef1010a7c4","resolution":{"observed_at":"2026-07-04T09:09:43.565648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.12346","last_updated":"2025-05-18T10:20:59Z","snapshot_observed_at":"2026-07-31T14:59:07.628754Z","submitted_at":"2025-05-18T10:20:59Z","title":"SEED-GRPO: Semantic Entropy Enhanced GRPO for Uncertainty-Aware Policy Optimization","version":1},"cited_work":{"arxiv_id":"2505.12346","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.12346","snapshot_observed_at":"2026-07-04T13:49:52.427933Z","title":"arXiv preprint arXiv:2505.12346 , year=","venue":null,"work_id":"1a5aa60e-eb9c-41b1-968f-cfa8336af063","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2505.12346","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:78d5d1ba23222a5b158e716b48a33f12295684f500a46ffd37f3657207163f21","observation_id":"14275d2a-83f5-485b-b2f0-ea9c0e1c32d6","resolution":{"observed_at":"2026-07-04T09:09:43.560962Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.23564","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:09:43.530437Z","title":"Segment policy optimization: Ef- fective segment-level credit assignment in rl for large language models","venue":null,"work_id":"c65cb5ef-2355-4a2d-8c67-89582e550b43","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:3469052e5b8f2dd917adc9d873db33e27e4042a72a9d1c617606b6ec236b6706","observation_id":"3a096dc8-2bbe-4dce-b337-98b70e7851a6","resolution":{"observed_at":"2026-07-04T09:09:43.531992Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.07017","last_updated":"2025-07-09T16:45:48Z","snapshot_observed_at":"2026-07-06T21:54:34.271025Z","submitted_at":"2025-07-09T16:45:48Z","title":"First Return, Entropy-Eliciting Explore","version":1},"cited_work":{"arxiv_id":"2507.07017","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.07017","snapshot_observed_at":"2026-07-04T09:09:43.539228Z","title":"First return, entropy-eliciting explore.CoRR, abs/2507.07017","venue":null,"work_id":"08bca174-6da3-4a11-adb1-c048a8af61d4","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2507.07017","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:e04d6c5a52659bd62f0aeb06eeaa4dad8acb9b98250f66409c3666818fd3476c","observation_id":"855d224b-6a44-4a97-ae37-fd7df83fc914","resolution":{"observed_at":"2026-07-04T09:09:43.540812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14286","last_updated":"2025-03-19T14:25:30Z","snapshot_observed_at":"2026-08-06T02:17:39.954909Z","submitted_at":"2025-03-18T14:23:37Z","title":"Tapered Off-Policy REINFORCE: Stable and efficient reinforcement learning for LLMs","version":2},"cited_work":{"arxiv_id":"2503.14286","doi":"10.48550/arxiv.2503.14286","metadata_source":"pith","pith_arxiv_id":"2503.14286","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tapered off-policy REINFORCE: Stable and efficient reinforcement learning for LLMs","venue":"cs.LG","work_id":"885b5bf5-7859-4e5a-af07-52a7da4f25c1","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2503.14286","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:0bb5409e1eb3bdc595584dd1e78ea08d8e998b224cc4496951f288b1ec6327e9","observation_id":"f39d511a-da72-4122-b12f-9f83743d30b5","resolution":{"observed_at":"2026-07-04T09:09:43.534766Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13585","last_updated":"2025-06-16T15:08:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-16T15:08:02Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","version":1},"cited_work":{"arxiv_id":"2506.13585","doi":"10.1109/tkde.2022.3168611","metadata_source":"pith","pith_arxiv_id":"2506.13585","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","venue":"cs.CL","work_id":"c59fbe20-f41e-4140-a81c-40a12e7e8364","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2506.13585","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:a962c3dc3ca2ec493b23758ab4e5eefb3aee79650789973d7178982bf4203683","observation_id":"dad12c66-8e34-44b0-808b-90fa3ac67f1a","resolution":{"observed_at":"2026-07-04T09:09:43.527440Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.07629","doi":"10.48550/arxiv.2508.07629","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Klear-reasoner: Advancing reasoning capability via gradient-preserving clipping policy optimization.arXiv preprint arXiv:2508.07629","venue":"ArXiv.org","work_id":"2d3fe64a-7a53-4e46-b832-c2a7bb443af1","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:0e5943639d5e9fbc388ca7420ce4128db8b57caf8c48f4b33d5b8db1146c8977","observation_id":"6e09a0a3-3f57-4e4a-8eab-93ad19dd7e0b","resolution":{"observed_at":"2026-07-04T09:09:43.533016Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.15778","last_updated":"2026-05-15T04:36:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T16:34:01Z","title":"Stabilizing Knowledge, Promoting Reasoning: Dual-Token Constraints for RLVR","version":2},"cited_work":{"arxiv_id":"2507.15778","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.15778","snapshot_observed_at":"2026-07-04T16:19:57.768398Z","title":"Stabilizing Knowledge, Promoting Reasoning: Dual-Token Constraints for RLVR","venue":"cs.CL","work_id":"b5286fd9-b359-49d2-a6bc-e370a3293e23","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2507.15778","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:ee687dd9592a7262b5ea69fde2ab4d82374437670ab8837efb13743aba4e90fe","observation_id":"ba3141db-5b46-46ac-8b82-e2539b502163","resolution":{"observed_at":"2026-07-04T09:09:43.600636Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.01347","doi":"10.48550/arxiv.2506.01347","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The surprising effectiveness of negative reinforcement in llm reasoning.arXiv preprint arXiv:2506.01347","venue":"ArXiv.org","work_id":"90fa4c77-dcfb-4253-827e-9bde0d342fcb","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:42a0143e99459e679caf2adadbc9505c9ab1d23070d6c3fb3a231bbb237c99d0","observation_id":"5f26c37a-d901-47b6-8880-98d8220ec7c4","resolution":{"observed_at":"2026-07-04T09:09:43.595453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11343","last_updated":"2025-06-12T06:03:24Z","snapshot_observed_at":"2026-07-06T21:09:50.780345Z","submitted_at":"2025-04-15T16:15:02Z","title":"A Minimalist Approach to LLM Reasoning: from Rejection Sampling to Reinforce","version":2},"cited_work":{"arxiv_id":"2504.11343","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.11343","snapshot_observed_at":"2026-07-09T04:05:55.456269Z","title":"A minimalist approach to llm reasoning: from rejection sampling to reinforce.arXiv preprint arXiv:2504.11343","venue":"cs.LG","work_id":"90392907-b8b0-4771-9ac8-6cd50245b1dd","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2504.11343","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:72724428a765156367e8a2ee10d6543c0704f65c78dab19c137d6d79122e8967","observation_id":"7d607ce7-cda3-4fb0-9d51-f107ba4b1af5","resolution":{"observed_at":"2026-07-04T09:09:43.575477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14945","last_updated":"2025-06-22T00:18:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-21T08:09:13Z","title":"Learning to Reason under Off-Policy Guidance","version":5},"cited_work":{"arxiv_id":"2504.14945","doi":"10.48550/arxiv.2504.14945","metadata_source":"pith","pith_arxiv_id":"2504.14945","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning to Reason under Off-Policy Guidance","venue":"cs.LG","work_id":"4ebcdbe2-5000-4f58-a7e0-aa9ae381b684","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2504.14945","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:b8b6cb357e9bbb20771b5d1558d861f23ab604b6efa2ebd6e42ce5d9fcf2728d","observation_id":"bc31eff6-7769-4562-ad98-fd5267be0c35","resolution":{"observed_at":"2026-07-04T09:09:43.587637Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.07527","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:25:57.811550Z","title":"Learning what reinforcement learning can't: Interleaved online fine-tuning for hardest questions","venue":null,"work_id":"77400081-d7b1-4be9-88ff-c1082c417c04","year":2026},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:47914e835b29bf322c0e7e08c68b82f32c79d493614274d2d61dd4544f46d48c","observation_id":"0065986c-10e2-4a9b-aeb9-f431a9b40910","resolution":{"observed_at":"2026-07-04T09:09:43.603102Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.19767","last_updated":"2025-06-24T16:31:37Z","snapshot_observed_at":"2026-07-06T21:47:01.972372Z","submitted_at":"2025-06-24T16:31:37Z","title":"SRFT: A Single-Stage Method with Supervised and Reinforcement Fine-Tuning for Reasoning","version":1},"cited_work":{"arxiv_id":"2506.19767","doi":"10.48550/arxiv.2506.19767","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.19767","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2506.19767 , year=","venue":"ArXiv.org","work_id":"03d392b0-d6dc-44a8-bfdd-888ccbc9e68e","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2506.19767","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:6d8e27c3cc472e2841aff37c3830b4b0c61576d9434446192238446ae9135483","observation_id":"967da951-9120-4295-8a7c-68731a45b4d7","resolution":{"observed_at":"2026-07-04T09:09:43.590285Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.20520","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T14:29:54.941865Z","title":"Arnal, G","venue":null,"work_id":"c99c7f38-d477-4148-a04c-9dc276438a43","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:f89807d7af9ce224b28b8d560fe1cd48aeeae503e55f14d707df6d328479f82f","observation_id":"55e747a9-cc43-4b9b-89f2-d32d4ae255f3","resolution":{"observed_at":"2026-07-04T09:09:43.592878Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.08221","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:09:43.596767Z","title":"Part i: Tricks or traps? a deep dive into rl for llm reasoning","venue":null,"work_id":"e0435aed-9cb6-435b-bb39-63bfca8ed5bd","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:fac31cf32b4ccf7721bf2589a23a23abc58260d29affe09fddae94a0a3a576d9","observation_id":"88f54f18-df8a-458b-8cdb-cfc12a3284ed","resolution":{"observed_at":"2026-07-04T09:09:43.598162Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"June , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:62c57a004a99160e79a696b864aba4f206d7e1324786e2f0d09e0bff1ecb9434","observation_id":"1f61da7e-d60a-46ae-9b91-ecf5b51a6037","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:fbc4f60691db1b8fb7df441645778b8230748da4c7e71a64d4f610171ce78bfc","observation_id":"aa319e90-2330-4c63-b119-7e5b25f98221","resolution":{"observed_at":"2026-07-04T09:09:43.547401Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T10:24:53.245739Z","title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:8c29414bf1bf731d16f7589097b80e55558805c4e61f21e04713d3d0469343e2","observation_id":"c620a0c4-3449-4494-aefc-ff0d418fb689","resolution":{"observed_at":"2026-06-26T10:24:53.245739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2504.02546","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:09:43.573095Z","title":"arXiv preprint arXiv:2504.02546 , year=","venue":null,"work_id":"cdcdaf53-666f-4c7f-8a7f-df8e9b3a4c49","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:f8c033a7a5def8cbf044994e0f7fc338db7a4559d0cf0bb639824f0acc327873","observation_id":"4640e1dc-0d32-4345-816d-47db5b224200","resolution":{"observed_at":"2026-07-04T09:09:43.574548Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18071","last_updated":"2025-07-28T11:11:33Z","snapshot_observed_at":"2026-08-06T04:31:03.113409Z","submitted_at":"2025-07-24T03:50:32Z","title":"Group Sequence Policy Optimization","version":2},"cited_work":{"arxiv_id":"2507.18071","doi":"10.48550/arxiv.2507.18071","metadata_source":"pith","pith_arxiv_id":"2507.18071","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Group Sequence Policy Optimization","venue":"cs.LG","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","year":2025},"citing_paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-06-26T10:24:53.245739Z"},"links":{"cited_paper":"/paper/2507.18071","citing_paper":"/paper/2606.22570"},"observation_digest":"sha256:3a646bfb32694a7911b4311dc8cee05ecbf9d62c8602ef23d3d3f8048de92681","observation_id":"c7137edb-6139-4332-bbb6-caed3ca98a75","resolution":{"observed_at":"2026-07-04T09:09:43.580238Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.22570","last_updated":"2026-06-21T16:14:46Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-05T07:45:25.638269Z","submitted_at":"2026-06-21T16:14:46Z","title":"What are Key Factors for Updates in RL for LLM Reasoning?"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":24,"parse_uncertain":0,"unresolved":12,"verified_exact":9,"verified_fuzzy":0},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 0 inbound Pith citation observations for arXiv:2606.22570."}