{"as_of":"2026-08-08T20:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a610833ce732ab8be2a12dbc745d9e58ad5a3b7edcc22343977e957964522169","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:40:08.061948Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T11:29:29.074324Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04746","snapshot_observed_at":"2026-08-02T11:29:29.074324Z","title":"Multi-layer grpo: Enhancing reasoning and self-correction in large language models.arXiv preprint arXiv:2506.04746, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.14502","last_updated":"2026-07-23T02:50:21Z","snapshot_observed_at":"2026-08-08T16:37:02.317008Z","submitted_at":"2026-06-12T14:33:55Z","title":"From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI","version":2},"reference_index":116,"source":"pdf_text","source_observed_at":"2026-08-02T11:29:29.074324Z"},"links":{"cited_paper":"/paper/2506.04746","citing_paper":"/paper/2606.14502"},"observation_digest":"sha256:926ea3805ae3759055c538a077a9ae9c42d0e88f99aa0f008f225548e0720fd6","observation_id":"fe35ccc6-759e-4a2e-b59d-15c4c08ee3a1","resolution":{"observed_at":"2026-08-02T11:29:29.074324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04746","snapshot_observed_at":"2026-07-30T22:49:42.994634Z","title":"arXiv preprint arXiv:2506.04746 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.23420","last_updated":"2026-07-26T02:35:47Z","snapshot_observed_at":"2026-08-02T21:56:50.028793Z","submitted_at":"2026-07-26T02:35:47Z","title":"LA-RL: Label-Aware Self-Reflection for Reinforcement Learning in Information Extraction","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-07-30T22:49:42.994634Z"},"links":{"cited_paper":"/paper/2506.04746","citing_paper":"/paper/2607.23420"},"observation_digest":"sha256:29b73aaa279b6b4d78d6d44490ef371157133f98de54bd2a9890e52ec0fc8a7d","observation_id":"6f8be023-0867-428d-8c04-a4f53cf464fd","resolution":{"observed_at":"2026-07-30T22:49:42.994634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.04746/citation-record","integrity":"/paper/2506.04746/integrity","json":"/paper/2506.04746/citation-record.json","paper":"/paper/2506.04746"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T10:40:06.139524Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.139524Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:7d77b6733d32ada10b81ef9bd748d06a2e4150d61dc0fff0b6f2ba5adbf3b5ff","observation_id":"ab798cd3-cf67-49c3-b187-b70b5b019c84","resolution":{"observed_at":"2026-08-07T10:40:06.139524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T10:40:06.185507Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.185507Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:c1c4d31b672f7f13e8d9fa27bfb9e53de1fe7577e306b1eee9abbab1d6f958ef","observation_id":"6b1636c3-4548-463d-82af-fbc33ca6f9f3","resolution":{"observed_at":"2026-08-07T10:40:06.185507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:08.995110Z","title":null,"venue":null,"work_id":"5f203a46-43cd-4dc2-bfae-118ea1e803ec","year":2022},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.253357Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:fc35ae5f574cf26bb494870f0af089665f34643530892b5497ab000a11ce562d","observation_id":"0b6f6705-fe19-4efb-9329-762b5281f9e1","resolution":{"observed_at":"2026-08-07T10:40:09.135642Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14008","last_updated":"2024-06-06T13:19:44Z","snapshot_observed_at":"2026-08-03T03:39:09.398343Z","submitted_at":"2024-02-21T18:49:26Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14008","snapshot_observed_at":"2026-08-07T10:40:06.320189Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.320189Z"},"links":{"cited_paper":"/paper/2402.14008","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:93e70e1181f93261c9d6c89e8c26a7ae7a51772c3483a8ecd3d75246bd045214","observation_id":"ba1f84a0-4c2f-4da3-bdc4-3521f882f266","resolution":{"observed_at":"2026-08-07T10:40:06.320189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:06.380688Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.380688Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:62b2744302aac689d5fcb8e481c89803565d5efa16c1b8bd4b367fed6dcddfb1","observation_id":"c8d4b795-a3c2-4ee4-843b-f256d021add3","resolution":{"observed_at":"2026-08-07T10:40:06.380688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01798","last_updated":"2024-03-14T04:27:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-03T04:56:12Z","title":"Large Language Models Cannot Self-Correct Reasoning Yet","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01798","snapshot_observed_at":"2026-08-07T10:40:06.447800Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.447800Z"},"links":{"cited_paper":"/paper/2310.01798","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:52eaee2ce93a95c1f294fad2743dcebba4ad3f27c3a9d7b3577cd37f12dd5a06","observation_id":"c89b90d7-cd9b-4837-a7a6-408fc3933a32","resolution":{"observed_at":"2026-08-07T10:40:06.447800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12917","last_updated":"2024-10-04T17:28:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-19T17:16:21Z","title":"Training Language Models to Self-Correct via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12917","snapshot_observed_at":"2026-08-07T10:40:06.568303Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.568303Z"},"links":{"cited_paper":"/paper/2409.12917","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:20b8e3db8a79d9e2c7688d317bad3fc76d7a27a99acbc919ddfa77ea7dc5ad7e","observation_id":"64bf6571-ca46-47fb-af1f-de4e316902a8","resolution":{"observed_at":"2026-08-07T10:40:06.568303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:06.692922Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.692922Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:799088ed36a261fc7872bbfe04aec60966b79c2ed08f0043f0e4a2ee123d3d13","observation_id":"29ec0150-9ffb-45e1-a693-8ccdfb7c255a","resolution":{"observed_at":"2026-08-07T10:40:06.692922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.07871","last_updated":"2018-11-19T18:48:04Z","snapshot_observed_at":"2026-07-06T07:15:51.575438Z","submitted_at":"2018-11-19T18:48:04Z","title":"Scalable agent alignment via reward modeling: a research direction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.07871","snapshot_observed_at":"2026-08-07T10:40:06.800735Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.800735Z"},"links":{"cited_paper":"/paper/1811.07871","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:03651e7beae59857727a3b1d7f78d815b2d7683ff3880281597ec2b9faeb3cc4","observation_id":"853a6945-caa1-4574-bb72-90775fe66610","resolution":{"observed_at":"2026-08-07T10:40:06.800735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:08.644619Z","title":null,"venue":null,"work_id":"490b4bf2-7e9b-4a40-bcbe-04954dd459be","year":2022},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:06.911148Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:1067935008ceb8ddad61a2529df594c1d2d325036bcdfe35da111a21fcb96a21","observation_id":"b5c80abc-02c5-4f3e-8a92-2b0f1e56b5d6","resolution":{"observed_at":"2026-08-07T10:40:08.808434Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:08.363762Z","title":null,"venue":null,"work_id":"959ac993-76ef-4e33-9f4c-72c03c15a181","year":2024},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.035824Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:c541df17f7c5e60f38b8adebf20749119384113d7449e6e0f2086451551b44ff","observation_id":"0bef0996-19a8-4213-870b-8946eada23d3","resolution":{"observed_at":"2026-08-07T10:40:08.480800Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-07T10:40:07.148178Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.148178Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:e8f8eac792b1f6b56ed66a3a16cd7e4f9b360acbc08c5c2ddab8a831f477d36c","observation_id":"c4301d10-da22-495e-8b85-527491d0e9f1","resolution":{"observed_at":"2026-08-07T10:40:07.148178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:08.289777Z","title":null,"venue":null,"work_id":"3437f680-e709-4417-aaf7-3c60163e710b","year":2020},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.273974Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:c7fd9ea04abdaa6424f98881df36fd22a5d4c8c5f318771b991ea5de2ecb237c","observation_id":"193424a8-b370-4cb1-a1ae-fd8716f02399","resolution":{"observed_at":"2026-08-07T10:40:08.354921Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T10:40:07.331186Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.331186Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:b77e13d9908e222216212313fbab67d7a92233c544fe86c1cfb6cf763872fb60","observation_id":"68d06202-fbea-444a-bbc7-f76a46853251","resolution":{"observed_at":"2026-08-07T10:40:07.331186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08146","last_updated":"2024-10-10T17:31:23Z","snapshot_observed_at":"2026-08-02T06:08:09.777151Z","submitted_at":"2024-10-10T17:31:23Z","title":"Rewarding Progress: Scaling Automated Process Verifiers for LLM Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08146","snapshot_observed_at":"2026-08-07T10:40:07.387674Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.387674Z"},"links":{"cited_paper":"/paper/2410.08146","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:aeaad60b557cea0ce452762baaaf69e3247e5e83fe3eaa89160b1fad2a825896","observation_id":"7c3ecdb3-18ab-4240-b43d-d400d0aeffeb","resolution":{"observed_at":"2026-08-07T10:40:07.387674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T10:40:07.456498Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.456498Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:457e773fe5a28f6700019d57ec8dedd94b26a7a8a84c66311094709aea7091fa","observation_id":"f5ad3a7b-2e7f-4bb1-9c07-ad54bd22ccaf","resolution":{"observed_at":"2026-08-07T10:40:07.456498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:07.548585Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.548585Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:7008f677c9c9766fb989b7deaa7a207416cc0c7f8c33ee9eea35d0f081c44057","observation_id":"a660bc98-c428-490f-8323-56b45bdf5715","resolution":{"observed_at":"2026-08-07T10:40:07.548585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-07T10:40:07.604698Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.604698Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:73afd8d4c5d24da342916093800b7f43e3a3b151138d444e0d1f30abf9ae786f","observation_id":"baea495c-e9ec-4df9-80e3-68e6d80d0c3f","resolution":{"observed_at":"2026-08-07T10:40:07.604698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-07T10:40:07.664835Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.664835Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:4620d4f244f951520fc1b986821cbf56739e472da9f168f8dc31510725908b7a","observation_id":"4998d84f-6eec-4663-a3dd-345085d855e6","resolution":{"observed_at":"2026-08-07T10:40:07.664835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08935","last_updated":"2024-02-19T14:07:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-14T13:41:54Z","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08935","snapshot_observed_at":"2026-08-07T10:40:07.733765Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.733765Z"},"links":{"cited_paper":"/paper/2312.08935","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:e0110086c72f729517abb890b22f683c26d04ec44d0e3efdf79eb5819b608327","observation_id":"a2c895e6-e531-4eac-987d-fceec4977b1e","resolution":{"observed_at":"2026-08-07T10:40:07.733765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-07T10:40:07.799079Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.799079Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:23805ef47ceae7ba4a4e2719ca9d51b98461711ac18721c1e65edc84622b2a55","observation_id":"98430df0-679c-475b-be78-8a36753bf931","resolution":{"observed_at":"2026-08-07T10:40:07.799079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01981","last_updated":"2024-12-02T21:20:02Z","snapshot_observed_at":"2026-08-06T03:48:05.506841Z","submitted_at":"2024-12-02T21:20:02Z","title":"Free Process Rewards without Process Labels","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01981","snapshot_observed_at":"2026-08-07T10:40:07.886754Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.886754Z"},"links":{"cited_paper":"/paper/2412.01981","citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:6cd7403ceb5fe302855da5a448957a3c495d750865a308e1f2845beb0b7c3f48","observation_id":"c6288b23-6d7c-4ca5-9ea3-4f13dee11fe3","resolution":{"observed_at":"2026-08-07T10:40:07.886754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:07.990278Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:07.990278Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:be1a205e0524da0b24a75421f7aacb7e562e7fbb62bfce02ece6eecf57375310","observation_id":"34338a91-b7fb-4e61-85dc-faf337e5ba63","resolution":{"observed_at":"2026-08-07T10:40:07.990278Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:40:08.061948Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T10:40:08.061948Z"},"links":{"citing_paper":"/paper/2506.04746"},"observation_digest":"sha256:843f681000a516e3016931966ed4c5f8236cde2d403eae11e955539ad47c8282","observation_id":"ad824c78-4743-4b58-8b78-7ffe378e44a0","resolution":{"observed_at":"2026-08-07T10:40:08.061948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.04746","last_updated":"2025-06-05T08:27:34Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T10:31:33.471015Z","submitted_at":"2025-06-05T08:27:34Z","title":"Multi-Layer GRPO: Enhancing Reasoning and Self-Correction in Large Language Models"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 2 inbound Pith citation observations for arXiv:2506.04746."}