{"as_of":"2026-08-05T20:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f0fdaa80eeae563142d42a36c57619c499325368f7983fb51783bdedbed2ab6c","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T14:23:27.591640Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T23:34:01.106317Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"cited_work":{"arxiv_id":"2508.21365","doi":"10.48550/arxiv.2508.21365","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.21365","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Think in games: Learning to reason in games via reinforcement learning with large language models,","venue":"ArXiv.org","work_id":"84af43b1-3481-4db6-9574-7bece161ed5a","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":288,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2508.21365","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:d815aba946935a18b3aa235cbc7240e834114746e07a8344de1da3b0e77f4119","observation_id":"682ba848-7a4b-4e77-a5d3-b9c69669f7e6","resolution":{"observed_at":"2026-06-27T09:50:48.309200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.21365","snapshot_observed_at":"2026-08-04T23:34:01.106317Z","title":"arXiv preprint arXiv:2508.21365 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01652","last_updated":"2026-08-03T03:43:57Z","snapshot_observed_at":"2026-08-05T19:51:08.105225Z","submitted_at":"2026-08-03T03:43:57Z","title":"SyncPlan: Long-Horizon LLM Coordination with Explicit Synchronization and Adaptive Correction","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T23:34:01.106317Z"},"links":{"cited_paper":"/paper/2508.21365","citing_paper":"/paper/2608.01652"},"observation_digest":"sha256:7b92a829637b8eedb51427936cc98e37518b747998d47af7bf25ccb7d858e933","observation_id":"f1e7d76b-4a07-474e-8ef8-18cc3018c509","resolution":{"observed_at":"2026-08-04T23:34:01.106317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2508.21365/citation-record","integrity":"/paper/2508.21365/integrity","json":"/paper/2508.21365/citation-record.json","paper":"/paper/2508.21365"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.18139","last_updated":"2024-09-30T00:40:00Z","snapshot_observed_at":"2026-08-04T03:17:09.662093Z","submitted_at":"2024-02-28T08:02:14Z","title":"Cause and Effect: Can Large Language Models Truly Understand Causality?","version":3},"cited_work":{"arxiv_id":"2402.18139","doi":"10.48550/arxiv.2402.18139","metadata_source":"pith","pith_arxiv_id":"2402.18139","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"Cause and Effect: Can Large Language Models Truly Understand Causality?","venue":"cs.CL","work_id":"4fd48f01-b5cb-4e6c-bcad-6b4271d22318","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:23.843562Z"},"links":{"cited_paper":"/paper/2402.18139","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:56bc5e68e9f9c70ddd1bb21245ddab245ba347db03daed8117c5eb0e5722d0ce","observation_id":"3462f19f-6220-49a1-a503-049fa24058ce","resolution":{"observed_at":"2026-08-05T14:23:28.247417Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:23.929458Z","title":null,"venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:23.929458Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:a66bd016d47edbf31a92d78062115e2f77cc16664fdda771ccf029609dd6ac47","observation_id":"6039e9bf-ca8c-441f-97b9-581a2cd2d618","resolution":{"observed_at":"2026-08-05T14:23:23.929458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T14:23:24.020818Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.020818Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:af0bf867d6d4f4d2b8602477aa86a24b81480711d15b02bcc06a655d55336193","observation_id":"9b2c6e68-646a-43be-92f7-971cf3f8ff64","resolution":{"observed_at":"2026-08-05T14:23:24.020818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00316","last_updated":"2025-06-06T08:14:05Z","snapshot_observed_at":"2026-07-06T20:15:03.658292Z","submitted_at":"2024-12-31T07:20:32Z","title":"MapEval: A Map-Based Evaluation of Geo-Spatial Reasoning in Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00316","snapshot_observed_at":"2026-08-05T14:23:24.072935Z","title":"Mapeval: A map-based evaluation of geo-spatial reasoning in foundation models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.072935Z"},"links":{"cited_paper":"/paper/2501.00316","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:8d75ba0b98e205055718adcaa4cd6a5a550945852439a13f9b00e1fec0bf32b1","observation_id":"65eeae8c-36e2-48ce-8939-8ceb51a810c7","resolution":{"observed_at":"2026-08-05T14:23:24.072935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.patrec.2007.06.013","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:27.961783Z","title":"Bayeschess: A computer chess program based on bayesian networks","venue":null,"work_id":"f2949566-7044-482a-91a5-03d0a02fc0c7","year":2008},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.156489Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:95e5a92c5dc1921382522f639dd884047a58a9e312b5d6cd1b5b262c5d8390ae","observation_id":"7a428987-6583-4222-9603-fb9898fd35bc","resolution":{"observed_at":"2026-08-05T14:23:28.080495Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2018.28345","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:29.842234Z","title":"Font and Tobias Mahlmann","venue":null,"work_id":"36499384-e109-4541-a224-1b7bf3044b36","year":2019},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.207525Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:f29a60e20d9e4951e850d04dd6848d9a4c0e6adfcc642f0de5547c9d64917205","observation_id":"d4b144ae-8637-4860-883c-01ca553fd71a","resolution":{"observed_at":"2026-08-05T14:23:29.926244Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.17131","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:29.571722Z","title":"Enabling self-improving agents to learn at test time with human-in-the-loop guidance","venue":null,"work_id":"187121ff-cbb7-4f66-a256-cad4d13661bd","year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.271531Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:a4e98bab091369d82f97bec5f4faabe5dc1c72419ba489d0509ecb53e0063904","observation_id":"ce16c2fa-817c-4f36-9a6e-3bc437a6a1b5","resolution":{"observed_at":"2026-08-05T14:23:29.665817Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:24.347757Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.347757Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:180a348b38beeefe1d5384b051538bb47fc052d697a25f8a4f979d977882e99c","observation_id":"7bb921e6-e04c-48b4-83d7-f04eef122fb0","resolution":{"observed_at":"2026-08-05T14:23:24.347757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-05T14:23:24.434666Z","title":"Openrlhf: An easy-to-use, scalable and high-performance rlhf framework","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.434666Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:0cf7e07661f83f948fb079df396d37f447950f32a298496ec33a0ddf433b2e5c","observation_id":"712a2b16-a84f-48c1-95db-a3c78fcec02a","resolution":{"observed_at":"2026-08-05T14:23:24.434666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02039","last_updated":"2026-06-08T03:25:04Z","snapshot_observed_at":"2026-08-03T07:51:57.376334Z","submitted_at":"2024-04-02T15:34:18Z","title":"A Survey on Large Language Model-Based Game Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02039","snapshot_observed_at":"2026-08-05T14:23:24.473717Z","title":"A survey on large language model-based game agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.473717Z"},"links":{"cited_paper":"/paper/2404.02039","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:cd8db0d38a19b2747661907b43eaba83a2e011c278d0480f6a5534cdcebc0085","observation_id":"4d97786b-a523-4a9f-a8cb-623b7b38872f","resolution":{"observed_at":"2026-08-05T14:23:24.473717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01118","last_updated":"2024-04-02T15:46:35Z","snapshot_observed_at":"2026-08-03T14:03:21.708112Z","submitted_at":"2024-02-02T03:22:12Z","title":"PokeLLMon: A Human-Parity Agent for Pokemon Battles with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01118","snapshot_observed_at":"2026-08-05T14:23:24.541597Z","title":"Pokellmon: A human-parity agent for pokemon battles with large language models, 2024 c","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.541597Z"},"links":{"cited_paper":"/paper/2402.01118","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:93f5988fc91a653adeb7b51faa165160dfb27159bab9031a2d10d2453f669a3b","observation_id":"37e15f8e-9e20-42f3-b515-e38236208f3f","resolution":{"observed_at":"2026-08-05T14:23:24.541597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.823398Z","title":"C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models","venue":null,"work_id":"f965f9f5-fd02-427e-9bea-6eacf7746590","year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.602432Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:14d76ab001c44ae4b192f649542982219032cf57b677eece038d824cdf1a8d6c","observation_id":"b120537f-e788-4299-b64b-8bc3f74f543e","resolution":{"observed_at":"2026-08-05T14:23:30.920211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-08-05T14:23:24.667850Z","title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.667850Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:70bbd982a857f4d515c1c242993f563ad45f3e929de2127f0a1e04e2757d8780","observation_id":"0a47fa39-e61c-4bd8-90a1-6ee078b5988d","resolution":{"observed_at":"2026-08-05T14:23:24.667850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20213","last_updated":"2025-04-01T14:43:55Z","snapshot_observed_at":"2026-07-06T19:24:30.632401Z","submitted_at":"2024-09-30T11:48:11Z","title":"Mind the GAP: Glimpse-based Active Perception improves generalization and sample efficiency of visual reasoning","version":2},"cited_work":{"arxiv_id":"2409.20213","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.20213","snapshot_observed_at":"2026-08-05T14:23:29.328001Z","title":"Mind the GAP: Glimpse-based Active Perception improves generalization and sample efficiency of visual reasoning","venue":"cs.CV","work_id":"4262f029-97b9-4f90-b5e6-2a6e7493ffdd","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.726570Z"},"links":{"cited_paper":"/paper/2409.20213","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:13ffbcee63b0167b3fa0887a525ae3a170f9ba6f90ae81af3edd1b90a40cd6d2","observation_id":"b8d24579-c921-478d-a8ca-613a937f29ae","resolution":{"observed_at":"2026-08-05T14:23:29.386562Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.683941Z","title":"School chinese benchmark, 2018","venue":null,"work_id":"4d4d20d1-1e19-43bc-950f-d2194a7c8b99","year":2018},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.798984Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:85d1ee8f5d7805f9506adbaa8218bb7c6a78bffc70ef1872cfc88b1260ee0182","observation_id":"a7266d34-80bb-40ec-a956-b4b069ea27d6","resolution":{"observed_at":"2026-08-05T14:23:30.727816Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07316","last_updated":"2025-05-21T13:38:27Z","snapshot_observed_at":"2026-07-06T20:34:34.959285Z","submitted_at":"2025-02-11T07:26:50Z","title":"CodeI/O: Condensing Reasoning Patterns via Code Input-Output Prediction","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07316","snapshot_observed_at":"2026-08-05T14:23:24.911042Z","title":"Codei/o: Condensing reasoning patterns via code input-output prediction","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.911042Z"},"links":{"cited_paper":"/paper/2502.07316","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:0149feee3c2ed3817e044754ddeac7b52838ab08a373a751ef421493ea817c92","observation_id":"4d0d153b-c670-4065-85c6-183d48962969","resolution":{"observed_at":"2026-08-05T14:23:24.911042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:24.975898Z","title":"Simpo: Simple preference optimization with a reference-free reward","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:24.975898Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:b5ede67c20765f1fcbb4d0b1818ab2b16eed8d6ead9d2083d964ea9e93760f09","observation_id":"d0e6f647-06ac-422b-be93-43d35f0c9b2e","resolution":{"observed_at":"2026-08-05T14:23:24.975898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1312.5602","last_updated":"2013-12-19T16:00:08Z","snapshot_observed_at":"2026-07-06T03:31:23.521122Z","submitted_at":"2013-12-19T16:00:08Z","title":"Playing Atari with Deep Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1312.5602","snapshot_observed_at":"2026-08-05T14:23:25.071210Z","title":"Playing atari with deep reinforcement learning","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.071210Z"},"links":{"cited_paper":"/paper/1312.5602","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:dd8f835e111b8ea94308ba150341b547b4f2c5faeefb1d9c00ad6dabf235e02e","observation_id":"8fe91552-3333-4fac-b23d-823e54708096","resolution":{"observed_at":"2026-08-05T14:23:25.071210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2021.30495","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:29.059062Z","title":"Creating pro-level AI for a real-time fighting game using deep reinforcement learning","venue":null,"work_id":"0ef64497-6163-4874-b91d-11fd6c28cb77","year":2022},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.134829Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:7852c4314881b4a358f75bef5f0b84db1898b9202f82c82234dd96db01156726","observation_id":"2ad2ca66-1586-4690-931c-d94db8641ab5","resolution":{"observed_at":"2026-08-05T14:23:29.160079Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.518999Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, et al","venue":null,"work_id":"779fda87-dffb-4bcf-b5ad-b8dd8e821487","year":2022},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.311280Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:e160d7bc1333015b9a03e898cbcb4fd04eb954efd35b1f70a8c74a593338967b","observation_id":"30ed4bd9-ea79-4473-95b0-6d637aab7769","resolution":{"observed_at":"2026-08-05T14:23:30.589098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:25.379171Z","title":"Manning, Stefano Ermon, and Chelsea Finn","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.379171Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:bcf72588b1c2f2c1be22c8fe6be548d7f8f674dff3640f841d3c202934fc622d","observation_id":"cf1e4c3f-0971-4329-965f-2e2f5a9a7df6","resolution":{"observed_at":"2026-08-05T14:23:25.379171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T14:23:25.463390Z","title":"Proximal policy optimization algorithms","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.463390Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:ee42b183246e0dcda48d6852f80eb9a6513939b6ca135ac518ce06e6b814f4eb","observation_id":"95e7b3ed-3fcd-47db-89ce-a6f1e284b427","resolution":{"observed_at":"2026-08-05T14:23:25.463390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T14:23:25.587387Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.587387Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:4f92d416562e786b7471a3a91bf036aeb6f2415bf3f8f9ecfe8e29d1c1e9d56a","observation_id":"6ad11b99-f8f3-4a63-b322-a6221a23776a","resolution":{"observed_at":"2026-08-05T14:23:25.587387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-05T14:23:25.667622Z","title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.667622Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:9f77cc5eb06c593e0d052dd5a766c67ef976453be217d571503971aa7f3572aa","observation_id":"5678c3e7-8d6f-494f-b41c-ad52a791efca","resolution":{"observed_at":"2026-08-05T14:23:25.667622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:25.840989Z","title":"Mastering the game of go with deep neural networks and tree search","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:25.840989Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:e2d91e4046b80e60401cf27fc6b29c952da64fde1e9404d14604110d93a99430","observation_id":"c3e1d71b-d402-466c-a341-5baf90e30292","resolution":{"observed_at":"2026-08-05T14:23:25.840989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1207.1411","last_updated":"2012-07-04T16:22:47Z","snapshot_observed_at":"2026-07-06T02:51:20.159654Z","submitted_at":"2012-07-04T16:22:47Z","title":"Bayes' Bluff: Opponent Modelling in Poker","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1207.1411","snapshot_observed_at":"2026-08-05T14:23:26.017512Z","title":"Bayes' bluff: Opponent modelling in poker","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.017512Z"},"links":{"cited_paper":"/paper/1207.1411","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:9297f3ac434a29f8b1d02510287c28c697e7d3935ca21315304db12460f2fa8c","observation_id":"a01d5683-b46c-40b4-9454-fdbea02541dc","resolution":{"observed_at":"2026-08-05T14:23:26.017512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.336552Z","title":"Brown, Adam Santoro, Aditya Gupta, et al","venue":null,"work_id":"62cf0502-18fe-421b-8173-982156eb4b40","year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.131722Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:637c278c1a2ad9e2d7b9323027bb968ad8f01e2279591df4fc6423840947c754","observation_id":"2eb77aa2-f1af-4a39-8bb1-bc87abb3f955","resolution":{"observed_at":"2026-08-05T14:23:30.402090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19918","last_updated":"2026-05-07T16:58:35Z","snapshot_observed_at":"2026-08-02T05:40:27.766263Z","submitted_at":"2025-02-27T09:40:13Z","title":"Meta-Reasoner: Dynamic Guidance for Optimized Inference-time Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19918","snapshot_observed_at":"2026-08-05T14:23:26.207323Z","title":"Meta-reasoner: Dynamic guidance for optimized inference-time reasoning in large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.207323Z"},"links":{"cited_paper":"/paper/2502.19918","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:c88629f6582bd56541f8d0fdb233fc3732b57d24e224355ec3d27c2d8d9310f2","observation_id":"8c0b0d61-6407-4193-b951-8fee35869ffe","resolution":{"observed_at":"2026-08-05T14:23:26.207323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:26.320509Z","title":"Le, Ed H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.320509Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:851ee4ef35b83311dab145ff217070bc51abf6efdcb3679702b43492d8ad7478","observation_id":"ad54ce16-f332-43d9-82dd-ddde84dc1c39","resolution":{"observed_at":"2026-08-05T14:23:26.320509Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01275","last_updated":"2024-01-09T18:54:05Z","snapshot_observed_at":"2026-07-06T17:10:52.289224Z","submitted_at":"2024-01-02T16:20:40Z","title":"CharacterEval: A Chinese Benchmark for Role-Playing Conversational Agent Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01275","snapshot_observed_at":"2026-08-05T14:23:26.402477Z","title":"Charactereval: A chinese benchmark for role-playing conversational agent evaluation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.402477Z"},"links":{"cited_paper":"/paper/2401.01275","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:bb127a61520794644adde699c8f8577513b0ab9661710da2a4f2faaabc0df337","observation_id":"7cb96572-77c7-4a03-bca6-b78143d72d34","resolution":{"observed_at":"2026-08-05T14:23:26.402477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1708.04782","last_updated":"2017-08-16T06:20:52Z","snapshot_observed_at":"2026-07-06T05:55:27.598127Z","submitted_at":"2017-08-16T06:20:52Z","title":"StarCraft II: A New Challenge for Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1708.04782","snapshot_observed_at":"2026-08-05T14:23:26.486476Z","title":"Starcraft ii: A new challenge for reinforcement learning, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.486476Z"},"links":{"cited_paper":"/paper/1708.04782","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:47513c0d8ba0e78b0094378faf941a626f6a3266be9341006e0ef454386b5054","observation_id":"e1d6f15b-32c1-4415-95b7-cce9431820eb","resolution":{"observed_at":"2026-08-05T14:23:26.486476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16291","last_updated":"2023-10-19T16:27:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-25T17:46:38Z","title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16291","snapshot_observed_at":"2026-08-05T14:23:26.587398Z","title":"Voyager: An open-ended embodied agent with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.587398Z"},"links":{"cited_paper":"/paper/2305.16291","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:be39977181b7780701aedd4e78dc54b928c476e6dd51de882b21d07f9d1cf7ac","observation_id":"6ec25b2f-cfd9-4b6c-a1f6-3fa2eeb994c4","resolution":{"observed_at":"2026-08-05T14:23:26.587398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01560","last_updated":"2024-07-08T05:56:47Z","snapshot_observed_at":"2026-08-02T14:50:04.461434Z","submitted_at":"2023-02-03T06:06:27Z","title":"Describe, Explain, Plan and Select: Interactive Planning with Large Language Models Enables Open-World Multi-Task Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01560","snapshot_observed_at":"2026-08-05T14:23:26.680923Z","title":"Describe, explain, plan and select: Interactive planning with large language models enables open-world multi-task agents, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.680923Z"},"links":{"cited_paper":"/paper/2302.01560","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:9d0b9a81c41bfee8e296fe13804f21d749ca097598c5916ebfe7ff8b9fc8aa6b","observation_id":"6ef93714-881f-4b35-80d7-bf99c4f62fd2","resolution":{"observed_at":"2026-08-05T14:23:26.680923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03622","last_updated":"2024-10-23T07:20:26Z","snapshot_observed_at":"2026-08-01T23:41:06.358986Z","submitted_at":"2024-04-04T17:45:08Z","title":"Mind's Eye of LLMs: Visualization-of-Thought Elicits Spatial Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03622","snapshot_observed_at":"2026-08-05T14:23:26.747722Z","title":"Mind's eye of llms: Visualization-of-thought elicits spatial reasoning in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.747722Z"},"links":{"cited_paper":"/paper/2404.03622","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:d4e33ff035cab82c98404e3a4c9b8bfe63ac2c1a74ec1372c4620ef47c3e4f77","observation_id":"1f538ca5-df5c-4728-8f1e-c26e3ae4beff","resolution":{"observed_at":"2026-08-05T14:23:26.747722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.13356","last_updated":"2025-03-17T16:42:34Z","snapshot_observed_at":"2026-08-03T08:04:05.117045Z","submitted_at":"2025-03-17T16:42:34Z","title":"Agents Play Thousands of 3D Video Games","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.13356","snapshot_observed_at":"2026-08-05T14:23:26.797595Z","title":"Agents play thousands of 3d video games","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.797595Z"},"links":{"cited_paper":"/paper/2503.13356","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:3d248c0f952117b0f5518342ceecd5e5312a7f91022b6b48f0ce3c0d8ca19733","observation_id":"e9e3842a-54be-4d53-beb0-b66881e83778","resolution":{"observed_at":"2026-08-05T14:23:26.797595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.12530","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:28.496152Z","title":"Policy-to-language: Train llms to explain decisions with flow-matching generated rewards","venue":null,"work_id":"9a83da07-05eb-42bf-be78-b9f892334e2f","year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:26.982264Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:4f988510eab15dd8b48acd0420789c8ee082629ab36cd47e670ac730b7824ad4","observation_id":"5dbef2c2-1c83-443d-93f3-8b5e6f3c6e76","resolution":{"observed_at":"2026-08-05T14:23:28.595131Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.188680Z","title":"Mastering complex control in moba games with deep reinforcement learning","venue":null,"work_id":"1da67548-d32d-422c-b88e-f190ac901258","year":2020},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.199232Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:af6c145d23bead2145b80b8555df2188b55de1712d50dda0de8ad4f4feae5940","observation_id":"e6705ad7-3bbe-4ea4-a259-aaef503742f2","resolution":{"observed_at":"2026-08-05T14:23:30.251013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-demos.30","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:27.727072Z","title":"C har P oet: A C hinese classical poetry generation system based on token-free LLM","venue":null,"work_id":"9f257eaa-ff5a-41b6-9e25-a1c595e2498b","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.276998Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:0360d364c44f19e88fa16f0d06a04a72869fc4d9cc91e127ec0c30ae19aa67da","observation_id":"34de5102-2fc4-47ed-a428-202845f6c721","resolution":{"observed_at":"2026-08-05T14:23:27.806325Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T14:23:30.036058Z","title":"Training interactive agent in large fps game map with rule-enhanced reinforcement learning","venue":null,"work_id":"d0a8b54b-d910-4c23-ba80-e1b8ca0c5d45","year":2024},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.367053Z"},"links":{"citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:f064979ae9a9bd8814e25e6d98c6169c027a673d20b7615597ccf4ce55692f7f","observation_id":"5a7d02ff-d198-40fd-88b0-c455de45c09f","resolution":{"observed_at":"2026-08-05T14:23:30.097095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.11506","last_updated":"2020-10-09T01:36:35Z","snapshot_observed_at":"2026-07-06T09:58:22.699198Z","submitted_at":"2020-09-24T06:17:10Z","title":"Ape210K: A Large-Scale and Template-Rich Dataset of Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.11506","snapshot_observed_at":"2026-08-05T14:23:27.444486Z","title":"Ape210k: A large-scale and template-rich dataset of math word problems, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.444486Z"},"links":{"cited_paper":"/paper/2009.11506","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:c263dcade0cfe3fa40cb116915d4df4b50c944dc36d1b9b2e2be9edf547b8a13","observation_id":"0aeb653d-cc59-49f5-ac7c-a1e24532cf66","resolution":{"observed_at":"2026-08-05T14:23:27.444486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-05T14:23:27.530464Z","title":"Instruction-following evaluation for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.530464Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:a02f5786159c490170a028fc9f28f5cb9f74cbac35b727bd0a953a8530a77ff8","observation_id":"4007ec55-0812-434b-9b84-7b42004e4305","resolution":{"observed_at":"2026-08-05T14:23:27.530464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08328","last_updated":"2025-01-24T20:15:10Z","snapshot_observed_at":"2026-08-04T22:36:15.059845Z","submitted_at":"2025-01-14T18:59:03Z","title":"PokerBench: Training Large Language Models to become Professional Poker Players","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08328","snapshot_observed_at":"2026-08-05T14:23:27.591640Z","title":"Pokerbench: Training large language models to become professional poker players","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T14:23:27.591640Z"},"links":{"cited_paper":"/paper/2501.08328","citing_paper":"/paper/2508.21365"},"observation_digest":"sha256:d4f5fb0b7f92b26f0ea1a4b00ce2e98f7f19a74d8d3836b5a8a22b5dd59925ee","observation_id":"9f49b6f6-5ae9-4788-8807-a56fc5849d5e","resolution":{"observed_at":"2026-08-05T14:23:27.591640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2508.21365","last_updated":"2025-08-29T07:13:39Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-05T14:23:02.986676Z","submitted_at":"2025-08-29T07:13:39Z","title":"Think in Games: Learning to Reason in Games via Reinforcement Learning with Large Language Models"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":27,"verified_exact":6,"verified_fuzzy":6},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 2 inbound Pith citation observations for arXiv:2508.21365."}