{"as_of":"2026-08-15T02:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4024dc6f75c4da78cfffac266fe43d1763c3e0b70f78ec6537290b7dba27d286","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T20:29:03.988127Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.09271/citation-record","integrity":"/paper/2608.09271/integrity","json":"/paper/2608.09271/citation-record.json","paper":"/paper/2608.09271"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:05.150727Z","title":"Maximum a posteriori policy optimisation","venue":null,"work_id":"f8bae17b-445d-4f4e-8cb9-fb87a41aab97","year":2018},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.752867Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:c9e73e0f284b285758a46018cb933601d158a7ee99b67e6ec0bab18fd9c409d7","observation_id":"48b53655-8338-455c-9072-cc43d51e5d70","resolution":{"observed_at":"2026-08-11T20:29:05.154815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.758603Z","title":"On-policy distillation of language models: Learning from self-generated mistakes","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.758603Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:a3e97bd6448b6cdb8fb339b12ac60b666bd8be6ed0683618ef568e9332ffaff4","observation_id":"b55a5c17-6a6d-44f5-821f-3c6cfdbe5ed4","resolution":{"observed_at":"2026-08-11T20:29:03.758603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21667","last_updated":"2026-06-04T07:59:25Z","snapshot_observed_at":"2026-08-12T18:42:41.645444Z","submitted_at":"2025-11-26T18:42:52Z","title":"Escaping the Verifier: Learning to Reason via Demonstrations","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.21667","snapshot_observed_at":"2026-08-11T20:29:03.762979Z","title":"Escapingtheverifier: Learningtoreasonviademonstrations.arXivpreprintarXiv:2511.21667, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.762979Z"},"links":{"cited_paper":"/paper/2511.21667","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:50f755f5c5f85c5f052031c259a35f972aac78f7d836a3d1e5c2a362e2200603","observation_id":"525ca485-55f0-49a7-b7e6-8fad8d4166b5","resolution":{"observed_at":"2026-08-11T20:29:03.762979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13585","last_updated":"2025-06-16T15:08:02Z","snapshot_observed_at":"2026-08-13T22:06:45.477681Z","submitted_at":"2025-06-16T15:08:02Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13585","snapshot_observed_at":"2026-08-11T20:29:03.767752Z","title":"Minimax-m1: Scaling test-time compute efficiently with lightning attention.arXiv preprint arXiv:2506.13585, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.767752Z"},"links":{"cited_paper":"/paper/2506.13585","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:7ecefdfae5883b664239f15ad3ed0b5d2b144931e2652e905b0afd5a6e1100d7","observation_id":"9c081b2c-b1cc-43ae-826a-154f1b865aa2","resolution":{"observed_at":"2026-08-11T20:29:03.767752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T20:29:03.773626Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.773626Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:33e01a5d06d55704c192735a1c36889bc1725031f8342ed81c8d1bcb71149e7a","observation_id":"e21b2de8-8e72-4244-a467-cbf86dfbdcbb","resolution":{"observed_at":"2026-08-11T20:29:03.773626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.779243Z","title":"What is the objective of reasoning with reinforcement learning?arXiv preprint arXiv:2510.13651, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.779243Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:acbb8a2c3caf9f7b023d2c1262997cf2ae339289d444fac90adaea57ae1775f6","observation_id":"e92ab6a6-f12e-4c5c-bf29-15f1bc327298","resolution":{"observed_at":"2026-08-11T20:29:03.779243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.783941Z","title":"Imagenet: A large-scale hierarchical image database","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.783941Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:a9b23d2e5fc2fe801d66fa39f607a7cc0711cb69c57911985b4ef166b226b9c8","observation_id":"f13f1dd0-43c6-4c6d-9695-30ca959afb0a","resolution":{"observed_at":"2026-08-11T20:29:03.783941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:05.105127Z","title":"Openvlthinker: Complex vision-language reasoning via iterative sft-rl cycles","venue":null,"work_id":"d9bf6837-bd49-4687-9e17-61baa07414f7","year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.790434Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:cfe32fcc2150b9781565afbc4e899e9e51def44615bd1b747f92635e59de63d5","observation_id":"84f6e10f-4685-4e7f-8cc3-fb2d32f8e0f2","resolution":{"observed_at":"2026-08-11T20:29:05.111003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:05.088562Z","title":"Cold-startreinforcementlearningwithsoftmaxpolicygradient.AdvancesinNeuralInformation Processing Systems, 30, 2017","venue":null,"work_id":"b4fbff60-11d1-47dd-a595-0f743803b808","year":2017},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.794665Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:1525b25d39424fc74091ecca5e35f3d0a3ee877aa80d728fd48e49911b8bed64","observation_id":"20687639-1a63-447a-a10a-f63fbced52b2","resolution":{"observed_at":"2026-08-11T20:29:05.094454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04475","last_updated":"2025-03-10T09:27:03Z","snapshot_observed_at":"2026-07-06T17:56:23.317089Z","submitted_at":"2024-04-06T02:29:02Z","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04475","snapshot_observed_at":"2026-08-11T20:29:03.802470Z","title":"Length-controlled alpacaeval: A simple way to debias automatic evaluators.arXiv preprint arXiv:2404.04475, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.802470Z"},"links":{"cited_paper":"/paper/2404.04475","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:8dfac744e4ffb915da8a096af69e2ac6e188bc612296c8766d5b2b3b20322e11","observation_id":"26e738b6-6dd2-424a-8a0f-45140ae1f74a","resolution":{"observed_at":"2026-08-11T20:29:03.802470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04178","last_updated":"2025-06-05T02:21:52Z","snapshot_observed_at":"2026-08-09T02:57:31.005115Z","submitted_at":"2025-06-04T17:25:39Z","title":"OpenThoughts: Data Recipes for Reasoning Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04178","snapshot_observed_at":"2026-08-11T20:29:03.807561Z","title":"Openthoughts: Data recipes for reasoning models.arXiv preprint arXiv:2506.04178, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.807561Z"},"links":{"cited_paper":"/paper/2506.04178","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:d2adc25d9cf3cef4409d242dc052d543d3ef361617189850845856eac1454e19","observation_id":"7beda436-9439-4ca8-a003-b6981708edcb","resolution":{"observed_at":"2026-08-11T20:29:03.807561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:05.069602Z","title":"Learning to reason for long-form story generation","venue":null,"work_id":"2e5af416-89e9-47ac-bbb1-0768a3cbe717","year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.813234Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:9c51c8817d900915c23d3e2af64e6fabaca7a1352617b2c70292682f01f48bba","observation_id":"4a17ffd8-7427-4f11-9cf7-79adc4121cf3","resolution":{"observed_at":"2026-08-11T20:29:05.075028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:05.049693Z","title":"RLP: Reinforcement as a pretraining objective","venue":null,"work_id":"4539ac48-6b98-49ee-8efa-c5087cecafc6","year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.817892Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:c7b90e52b2be565eccc7c7feafcaa2dcc35c11e52d94ff3727cdfd3715636c72","observation_id":"48f9ec6e-9262-4274-afaa-6fb7501c6178","resolution":{"observed_at":"2026-08-11T20:29:05.054235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.823062Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.823062Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:61a98f0db3c2f8887267c8fccc18154003db17e52890d5c175a42e2b62dea12e","observation_id":"05494d81-0622-43c3-b6f8-6e8f374f2ce4","resolution":{"observed_at":"2026-08-11T20:29:03.823062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2604.02323","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.537950Z","title":"Shah, Qihua Dong, Zilin Xiao, Jaywon Koo, and Vicente Ordonez","venue":null,"work_id":"e1115849-98e9-461a-b825-97ee3b80d34a","year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.827802Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:995444bc2986ca8aa3c986c7d841b027249c2d88fac0644c7bd42609ebfd0b6b","observation_id":"6cfee7c8-6f4c-48d2-9d30-2b7b50364774","resolution":{"observed_at":"2026-08-11T20:29:04.545909Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:05.017500Z","title":"Deepmath-103k: A large-scale, challenging, decontaminated, and verifiable mathematical dataset for advancing reasoning","venue":null,"work_id":"8a80532a-4258-468d-8c05-472bab8af07a","year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.832205Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:45ad539faf32f75e4b13da4a0e8c018362a364946fdfd4fd604bf3fef75466e3","observation_id":"35b152db-d9c5-4e67-be3b-5f387a809020","resolution":{"observed_at":"2026-08-11T20:29:05.022024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.997999Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":"ce338c33-e5f6-4808-9c3c-516faeff86e5","year":2021},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.837328Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:c8607387d0fa25d901b3c68910558585acf279b877082f9a457d9649e741798f","observation_id":"9b185181-cef0-4488-85ec-28312844286e","resolution":{"observed_at":"2026-08-11T20:29:05.003190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.841893Z","title":"MeetingBank: A benchmark dataset for meeting summarization","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.841893Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:4cc01812c72e9f424b6f160d8a8027addc39bdb76afc8862d34b0f726ab24627","observation_id":"15b122f3-3271-47d1-a85e-1056e4c2a181","resolution":{"observed_at":"2026-08-11T20:29:03.841893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.10104","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.457313Z","title":"Answer-consistent chain-of-thought reinforcement learning for multi-modal large langauge models.arXiv preprint arXiv:2510.10104, 2025","venue":null,"work_id":"8cd7d376-df8c-4a17-b2b8-312957d2cbb0","year":null},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.846255Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:9e135b0368476ff956fccd9aa6ad06def1fbdcbde6377ecf9bfbfa80d31f3780","observation_id":"3c36242e-055c-4d68-baa6-1c03b07921da","resolution":{"observed_at":"2026-08-11T20:29:04.467803Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-11T20:29:03.849865Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.849865Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:ae14566a504f02eb2a05e6ed02981575e9a13ba0ca6c9e2f1521e116574d2153","observation_id":"795c7606-06d9-4595-9f60-48d13b5bff96","resolution":{"observed_at":"2026-08-11T20:29:03.849865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.977907Z","title":"Understanding r1- zero-like training: A critical perspective","venue":null,"work_id":"16d4eda7-fc23-4ee7-a989-36c29fc51680","year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.854986Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:3525ed262140a4c279f2092da05389426ed1dde82638d4c7c17da23a665d310b","observation_id":"aa3ffc40-e62f-4a9e-ba83-fd47bd5756da","resolution":{"observed_at":"2026-08-11T20:29:04.984900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-14T20:13:52.872565Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-11T20:29:03.858538Z","title":"Decoupled weight decay regularization.arXiv preprint arXiv:1711.05101, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.858538Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:a009d78c4be7023265dfdd0ec452cd24e77c543b14dc9367c4003cbb874a6f80","observation_id":"bbf87a8c-ddbe-4498-8be6-dde95ef68651","resolution":{"observed_at":"2026-08-11T20:29:03.858538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.861966Z","title":"On-policy distillation.Thinking Machines Lab: Connectionism, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.861966Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:5176b7625223e13efdac4b216be07f1b31c596b114842e8d9cf319cd9591667f","observation_id":"e4fd2436-2fc2-4385-b832-dbd69df5c953","resolution":{"observed_at":"2026-08-11T20:29:03.861966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14652","last_updated":"2025-06-09T17:40:25Z","snapshot_observed_at":"2026-08-12T18:09:06.281598Z","submitted_at":"2025-05-20T17:41:33Z","title":"General-Reasoner: Advancing LLM Reasoning Across All Domains","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14652","snapshot_observed_at":"2026-08-11T20:29:03.866650Z","title":"General-reasoner: Advancing llm reasoning across all domains.arXiv preprint arXiv:2505.14652, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.866650Z"},"links":{"cited_paper":"/paper/2505.14652","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:060fee9410a9efdb40b042b8dfb05cab453cac49432e4db9bc8c4cea13c449c5","observation_id":"52b88f80-ef63-4ff9-8573-7d2bbb375abc","resolution":{"observed_at":"2026-08-11T20:29:03.866650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.961815Z","title":"Reward augmented maximum likelihood for neural structured prediction","venue":null,"work_id":"cfaab53f-0959-421d-8ea8-787d676baa3d","year":2016},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.872377Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:f3b8e841d0e078ece85d12026db58f08bc845d1096091d672100dc8d1c33b4af","observation_id":"f8ec9b47-7531-485b-acc9-1ae53493c044","resolution":{"observed_at":"2026-08-11T20:29:04.967759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.945121Z","title":"Iterative reasoning preference optimization","venue":null,"work_id":"fb2b8b92-3031-4686-a340-2c0c3ba95289","year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.876731Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:d62c435b5131f5defb29deb534175bf49c7f42742c0741b0915f477257754d83","observation_id":"a1ef6423-ea6a-4ad1-a018-93043bdfcc5d","resolution":{"observed_at":"2026-08-11T20:29:04.951654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.04879","last_updated":"2026-05-26T12:42:28Z","snapshot_observed_at":"2026-08-14T23:42:11.859761Z","submitted_at":"2026-02-04T18:59:04Z","title":"Rethinking the Trust Region in LLM Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.04879","snapshot_observed_at":"2026-08-11T20:29:03.881817Z","title":"Rethinking the trust region in llm reinforcement learning.arXiv preprint arXiv:2602.04879, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.881817Z"},"links":{"cited_paper":"/paper/2602.04879","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:639180e2be0b89492ef0c47609c47ae5ffb1d64eac22e09f606a222330e858f8","observation_id":"51f6a1b5-835f-47b5-8420-966b7dbfdea3","resolution":{"observed_at":"2026-08-11T20:29:03.881817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.889681Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.889681Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:a187e9b5853ad25a430417d5f62eabf5b34e835f96cd3555405cb37cfe36ab4b","observation_id":"55fe61b9-3264-4cfd-b253-47e751c87af8","resolution":{"observed_at":"2026-08-11T20:29:03.889681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.905706Z","title":"Direct preference optimization: Your language model is secretly a reward model","venue":null,"work_id":"a17790d0-82ed-4f2f-8ad8-f524f7e7187f","year":2023},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.900830Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:cd1c1369cfce43154250a92c501bb46ef675f5a76fff914fb86b5d8daba05621","observation_id":"179e6939-895e-4b35-80e7-2111e926cea2","resolution":{"observed_at":"2026-08-11T20:29:04.911990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.885631Z","title":null,"venue":null,"work_id":"d9bfffd3-d172-41af-a17a-a244f9f656e1","year":null},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.905739Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:ded3f0b3665376b5363a6b1aa970bc44d06a927110905e3cf43776e527bf79cb","observation_id":"3fe1534d-def2-40eb-a0e7-b2d07c77d067","resolution":{"observed_at":"2026-08-11T20:29:04.890952Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.852400Z","title":"Optimal completion distillation for sequence learning","venue":null,"work_id":"150d5003-9136-4fec-8867-faa7fce96ca3","year":2019},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.916418Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:bc66e0ecd62ee482ac665d53af515d6ad36aa8c6fa1c048bc753598f2d5f272b","observation_id":"453e334e-fed7-4919-985c-4d93a7cf781a","resolution":{"observed_at":"2026-08-11T20:29:04.856233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-11T20:29:03.921502Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.921502Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:4db2f0a9933348d2f0ffb5f701e4a0793fadca6fdc272f8dcbe191905dd118fc","observation_id":"c449c312-c2ba-4e89-8204-79e2a2a4606b","resolution":{"observed_at":"2026-08-11T20:29:03.921502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-11T20:29:03.926196Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models.arXiv preprint arXiv:2402.03300, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.926196Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:244d8c8a17398158d72e1c32a1a1003f018917cd1d27cde2e50ba45a5dbbf1d3","observation_id":"8f039705-4656-4b94-8dfe-0fa5e4a2df2b","resolution":{"observed_at":"2026-08-11T20:29:03.926196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.19256","snapshot_observed_at":"2026-08-11T20:29:03.931344Z","title":"Hybridflow: A flexible and efficient rlhf framework.arXiv preprint arXiv: 2409.19256, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.931344Z"},"links":{"cited_paper":"/paper/2409.19256","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:d95dbd6ecf9a665f4b6cca9634c7ed2e8402482690354a1205319350764ee49e","observation_id":"a0f34d5e-7215-45a8-a781-3358ef13b455","resolution":{"observed_at":"2026-08-11T20:29:03.931344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-08-10T14:48:37.196526Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-11T20:29:03.935373Z","title":"Crossing the reward bridge: Expanding rl with verifiable rewards across diverse domains.arXiv preprint arXiv:2503.23829, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.935373Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:16e42a15195a7a66d242d511d0085807c40b644685179e55024242b308e43762","observation_id":"c91d0b00-6ba4-407d-9136-bf043f5db81e","resolution":{"observed_at":"2026-08-11T20:29:03.935373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.940136Z","title":"Maximum likelihood reinforcement learning.arXiv preprint arXiv:2602.02710, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.940136Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:086535d09c9a6d83a11e94a5572c1ece6a10829456ffb36591782003e2ea43eb","observation_id":"d1e7ee22-279d-4e3f-b327-ce76786d6063","resolution":{"observed_at":"2026-08-11T20:29:03.940136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.840124Z","title":"Vl-rethinker: Incentivizing self- reflection of vision-language models with reinforcement learning","venue":null,"work_id":"42564778-c6c2-452f-90f7-82ff18fcd3a0","year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.945359Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:160c6285c2035a8e00e5d67469113d94fedc61f27a711a72640b60c6e962beb4","observation_id":"c3f0d3d6-0693-4bf6-9d57-009fef5ea153","resolution":{"observed_at":"2026-08-11T20:29:04.844077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.822864Z","title":"Sota with less: Mcts-guided sample selection for data-efficient visual reasoning self-improvement","venue":null,"work_id":"89c0ebbf-5d73-49f7-bcf0-618cbee2aaba","year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.949476Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:df0ccdbecf6be2dd95b29ff15259e4f2cd0b46e3f30850afbe97c3c75c10589b","observation_id":"5b43ec2a-b9e5-4ed5-a44e-8c85cc8456b3","resolution":{"observed_at":"2026-08-11T20:29:04.830899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.807344Z","title":"Thoughts are all over the place: On the underthinking of long reasoning models","venue":null,"work_id":"c9f528d4-699d-466c-996b-93b6ce28449c","year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.953507Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:fb2885cbb2967048c3d235124e9108852c3def6348509c812c64e34bc5052751","observation_id":"002ece37-9b4a-47f5-9574-3e38074cefb9","resolution":{"observed_at":"2026-08-11T20:29:04.812235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.957406Z","title":"Sportr: A benchmark for multimodal large language model reasoning in sports, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.957406Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:302bfe03226f6e3435a7db4836b8044ab40a89c654c9062fea59fe0e094bd893","observation_id":"5eaaa617-1b79-453c-91ba-0f1d303e262f","resolution":{"observed_at":"2026-08-11T20:29:03.957406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.792397Z","title":"Proxythinker: Test-time guidance through small visual reasoners","venue":null,"work_id":"ec9e7203-a065-4a9e-bc2b-3d4ef2b7a54d","year":2026},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.961440Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:6e64295a9ce246dd6efafa9b0eac217b67d065b74978f0f5dda7e210d09175fd","observation_id":"d40d4dbe-5ab3-4b13-98ad-cca8b8785645","resolution":{"observed_at":"2026-08-11T20:29:04.797570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-11T20:29:03.965787Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.965787Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:72ad7c94b9b3a2d4e40ceb3a7f009dedfb41cf3b4e2526608e78f7321c60509f","observation_id":"2e12425e-c792-4fb2-82e7-bcae2d64769a","resolution":{"observed_at":"2026-08-11T20:29:03.965787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.970117Z","title":"DAPO: An open-source LLM reinforcement learning system at scale","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.970117Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:49f2e3b21268cc5e56bc06c9ee94facd5c567ec73d3fcb4af1d95b2bb5c5d0f9","observation_id":"b3a193c2-0a9a-41db-833d-8db2ba8ccc8a","resolution":{"observed_at":"2026-08-11T20:29:03.970117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:03.974838Z","title":"STar: Bootstrapping reasoning with reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.974838Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:6f5a7973ae8f67012a2a820ffedd974adb9c15be58b6c0e51936020a50c45739","observation_id":"0d210592-509d-4f11-8a92-9c657cc31ef6","resolution":{"observed_at":"2026-08-11T20:29:03.974838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08297","last_updated":"2025-07-21T10:37:40Z","snapshot_observed_at":"2026-08-07T21:25:45.428255Z","submitted_at":"2025-07-11T04:07:10Z","title":"KAT-V1: Kwai-AutoThink Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.08297","snapshot_observed_at":"2026-08-11T20:29:03.979079Z","title":"Kat-v1: Kwai-autothink technical report.arXiv preprint arXiv:2507.08297, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.979079Z"},"links":{"cited_paper":"/paper/2507.08297","citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:4ac9c870fccdc9c457f7fa36c5d669c8a5c92bcef28af6af67ae4fa71bba0fa2","observation_id":"b00b7bd6-e5af-4951-b3f8-389e1fe0f17d","resolution":{"observed_at":"2026-08-11T20:29:03.979079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.753689Z","title":"Reinforcing general reasoning without verifiers","venue":null,"work_id":"da0e0257-877c-42fe-b469-2659642993a8","year":null},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.983724Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:84f18fbbeecdb847e0fec362f99224dbfbd08325969fbd56873e7d28a32281fb","observation_id":"7b41180f-ec8e-476d-ad7e-6b1f00cae5c5","resolution":{"observed_at":"2026-08-11T20:29:04.757871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.866807Z","title":null,"venue":null,"work_id":"a4bca34b-feaa-476f-80b7-90be0a6a9447","year":null},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.911135Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:3fbb1ef75c43f72293d14d6e9d9270e239f80e45a6f3cb6c7cf3540dec25285a","observation_id":"bb7d2340-b2b4-46a5-866f-c891faa92bc3","resolution":{"observed_at":"2026-08-11T20:29:04.871224Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T20:29:04.736954Z","title":"Please reason step by step, and put your final answer within \\boxed{}","venue":null,"work_id":"171ea657-a4d2-4d90-ac80-48c7fd0cd9fb","year":null},"citing_paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-11T20:29:03.988127Z"},"links":{"citing_paper":"/paper/2608.09271"},"observation_digest":"sha256:2994867a2f0af7ec7d322f2567091f05c96734b94ece3700092231a1a72be9d0","observation_id":"0ab855d8-2657-4615-997b-dada58adeac7","resolution":{"observed_at":"2026-08-11T20:29:04.742339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.09271","last_updated":"2026-08-10T08:27:15Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T19:37:44.330209Z","submitted_at":"2026-08-10T08:27:15Z","title":"SoftmaxGRPO: Learning to Reason using Softmax Advantage Group Estimation"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":28,"verified_exact":2,"verified_fuzzy":18},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 0 inbound Pith citation observations for arXiv:2608.09271."}