{"as_of":"2026-08-11T19:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:319b4c1373f29dafe9787b94a3c2747bc0def2e8e0929b636635b7bf50a96aa2","coverage":[{"denominator":115,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:27:03.187728Z","state":"measured"},{"denominator":122,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":122,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":22,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T22:50:33.015304Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-07T22:50:33.015304Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.09053","last_updated":"2025-08-05T02:23:31Z","snapshot_observed_at":"2026-08-08T10:12:47.586452Z","submitted_at":"2025-02-13T08:08:27Z","title":"Game Theory Meets Large Language Models: A Systematic Survey with Taxonomy and New Frontiers","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T22:50:33.015304Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2502.09053"},"observation_digest":"sha256:6cd394cbe2aeb3bda09efb63b957e357a738994f896fbb72050c7f580d67eb26","observation_id":"6eff8939-1f26-465a-9aab-d505c04f2523","resolution":{"observed_at":"2026-08-07T22:50:33.015304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-06T04:33:42.500139Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.03368","last_updated":"2025-08-18T09:53:16Z","snapshot_observed_at":"2026-08-07T21:23:20.919037Z","submitted_at":"2025-08-05T12:15:59Z","title":"Game Reasoning Arena: A Framework and Benchmark for Assessing Reasoning Capabilities of Large Language Models via Game Play","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T04:33:42.500139Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2508.03368"},"observation_digest":"sha256:3482d7075efac449a77e6076e4a672fbda7cfc4cafc53fde44737c6e64bb9a0a","observation_id":"6b6def62-bd00-4966-905d-2beeba9e4b8f","resolution":{"observed_at":"2026-08-06T04:33:42.500139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2509.02544","last_updated":"2025-09-05T14:59:27Z","snapshot_observed_at":"2026-08-11T12:35:47.937091Z","submitted_at":"2025-09-02T17:44:45Z","title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T10:13:58.774968Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2509.02544"},"observation_digest":"sha256:8870bb08da98810c11a2b79deac783f531c0a290d449188eff4ef25731ddecd4","observation_id":"c87a4664-3ed6-421e-bf53-762d56b3599c","resolution":{"observed_at":"2026-05-13T10:13:59.184183Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":203,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:cf2ab83f39a15791d4b9f498f5b35c3a0fd7140d2708fff144f33c1c3b4387e0","observation_id":"27ec31b8-d11b-4952-889c-a66ceb5987ad","resolution":{"observed_at":"2026-05-18T00:05:31.526497Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2603.15432","last_updated":"2026-04-08T15:52:59Z","snapshot_observed_at":"2026-08-11T13:24:18.985998Z","submitted_at":"2026-03-16T15:37:07Z","title":"Gym-V: A Unified Vision Environment System for Agentic Vision Research","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T10:05:09.049846Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2603.15432"},"observation_digest":"sha256:c622086ba9d200f611355d2f884c1008342aca0ed8020e74a68a8c68c0fba6dd","observation_id":"f8567ab9-4edc-49b5-a74d-036f1e741261","resolution":{"observed_at":"2026-05-15T10:05:26.044132Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2604.20043","last_updated":"2026-04-21T22:55:57Z","snapshot_observed_at":"2026-08-10T21:52:27.129411Z","submitted_at":"2026-04-21T22:55:57Z","title":"TriEx: A Game-based Tri-View Framework for Explaining Internal Reasoning in Multi-Agent LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T02:04:57.744754Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2604.20043"},"observation_digest":"sha256:2e8ce9989c2da6e3d5ee14cd44b2b19ae2e125ad5f4cd770cce0c3500ec8b90d","observation_id":"5f80c1c4-4c08-452c-a764-51911f301c29","resolution":{"observed_at":"2026-05-11T13:16:06.152542Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2604.20987","last_updated":"2026-04-22T18:17:17Z","snapshot_observed_at":"2026-08-11T10:29:19.336072Z","submitted_at":"2026-04-22T18:17:17Z","title":"Co-Evolving LLM Decision and Skill Bank Agents for Long-Horizon Tasks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T00:14:07.017420Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2604.20987"},"observation_digest":"sha256:bfe8a3d0c5354baa9e58004ab4fa53839cd04405bb6c5bea261d9df188a90382","observation_id":"e18a66e5-bba2-4612-b96a-11391a6046d7","resolution":{"observed_at":"2026-05-10T00:14:46.428384Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.00347","last_updated":"2026-05-01T02:05:56Z","snapshot_observed_at":"2026-08-11T00:38:42.624588Z","submitted_at":"2026-05-01T02:05:56Z","title":"Odysseus: Scaling VLMs to 100+ Turn Decision-Making in Games via Reinforcement Learning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-09T20:22:58.061772Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.00347"},"observation_digest":"sha256:f2d3175e7bac873289bb337adb16a652e5dbe6f54a87c670242bfbcfa676b0aa","observation_id":"1fb4d952-c576-432d-96bc-4165ca85794e","resolution":{"observed_at":"2026-05-11T15:16:09.364638Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-14T19:05:36.511150Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:e9316ad5023f2e0b2b0793912664dc4af1e4453a9c85a2bdb6e2393223312e03","observation_id":"70f766f4-694a-44b7-b27c-a41a16861d3c","resolution":{"observed_at":"2026-05-14T19:07:51.419743Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T05:59:44.669877Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:569ab0e04dddb860016e9f397315103b181b10c1f17e71659ca6530a97a3041a","observation_id":"5d47647a-68a4-4f89-984d-0df532ab97c7","resolution":{"observed_at":"2026-05-15T05:59:48.141281Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T21:31:20.079403Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:45c11ed76ff4ff7da4ad1c29355c532f735000a1aa53dd55b5f385c7ba381bb0","observation_id":"1d368f40-3266-41ea-8cdd-8b9fb6fa95df","resolution":{"observed_at":"2026-06-30T21:35:04.559898Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T12:24:06.062957Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:5194faf5aad413cdd686f62ad03648a4924b8e438fa107ba2697e6c37045a5cb","observation_id":"cfd7b7b2-11a5-41eb-8577-10df4fcd792b","resolution":{"observed_at":"2026-05-20T12:28:17.261880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-25T05:45:04.573722Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:192190a0b7f4a5e827b4b4a91c5e3fa376feabcc5dc5dcc95eb65869460303c6","observation_id":"2049d76d-8774-4384-a78f-ba5c6d631ed4","resolution":{"observed_at":"2026-05-25T05:45:23.221897Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.29512","last_updated":"2026-05-28T07:33:47Z","snapshot_observed_at":"2026-07-06T23:38:56.014434Z","submitted_at":"2026-05-28T07:33:47Z","title":"MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T07:15:27.939886Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.29512"},"observation_digest":"sha256:4041afc7617dc3f51964e7cd079f4b6d03889bb61cc06172aa0b78a6f8544acb","observation_id":"cb5202f0-d7ec-4fae-b9ff-6d58f9651a63","resolution":{"observed_at":"2026-06-29T07:23:13.196706Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2605.29653","last_updated":"2026-05-28T09:16:22Z","snapshot_observed_at":"2026-07-06T23:39:01.234601Z","submitted_at":"2026-05-28T09:16:22Z","title":"PTCG-Bench: Can LLM Agents Master Pok\\'emon Trading Card Game?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T07:21:49.763994Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2605.29653"},"observation_digest":"sha256:4f7dd31d05065dc36be457ac2c61ffc46cbeee721c82bfc4d8a1449c365de822","observation_id":"23f86099-b0a4-427e-800a-d1b8d9bdeb5f","resolution":{"observed_at":"2026-06-29T07:23:12.593221Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.06556","last_updated":"2026-06-04T10:43:14Z","snapshot_observed_at":"2026-08-09T06:57:25.072034Z","submitted_at":"2026-06-04T10:43:14Z","title":"Robots Need More than VLA and World Models","version":1},"reference_index":152,"source":"arxiv_source","source_observed_at":"2026-06-28T01:01:33.530167Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.06556"},"observation_digest":"sha256:f90feca4b127bf99d09daf28a20b1d6c1b2ae8a83f8d42ca1dcaba070229c4b6","observation_id":"cad8132a-3e76-4034-aed2-e4abc28a1d66","resolution":{"observed_at":"2026-06-28T01:11:28.899789Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.09826","last_updated":"2026-06-08T17:59:43Z","snapshot_observed_at":"2026-08-02T20:00:31.322130Z","submitted_at":"2026-06-08T17:59:43Z","title":"OmniGameArena: A Unified UE5 Benchmark for VLM Game Agents with Improvement Dynamics","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-06-27T16:50:36.194650Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.09826"},"observation_digest":"sha256:efa8dc9a86c8d1cabf7be6254753068ab89428a1382b002294b987944f10be81","observation_id":"f4c3ebe9-e250-4a38-9fe2-063b44fa6a8c","resolution":{"observed_at":"2026-07-03T01:07:29.947294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":116,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:2e879c4f6e22d4dc13e388fb0db1c7859836bb91f9af200a6f1d2e587de2c9f3","observation_id":"09f90fa0-57f8-4c7f-9529-c4a8dd5f4cc1","resolution":{"observed_at":"2026-06-27T09:50:48.495612Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.18950","last_updated":"2026-06-18T08:25:27Z","snapshot_observed_at":"2026-08-07T06:04:46.021695Z","submitted_at":"2026-06-17T11:32:51Z","title":"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-26T21:02:26.081166Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.18950"},"observation_digest":"sha256:6137bd8327e6542e595792890113b6cff6f00094918f05087945098fd0160f22","observation_id":"0be5c38e-8540-42a0-be0c-76e3c213b0e2","resolution":{"observed_at":"2026-07-04T00:39:17.299890Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.19338","last_updated":"2026-06-17T17:59:34Z","snapshot_observed_at":"2026-07-06T23:54:37.596257Z","submitted_at":"2026-06-17T17:59:34Z","title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-26T21:17:02.332687Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.19338"},"observation_digest":"sha256:866489f2bbe55cc352cca6141eacf71c07a34a46f4a9eba88cd8ea48bcf30bc3","observation_id":"914f01fc-110b-4bcd-9f5a-dbb5aad4ede8","resolution":{"observed_at":"2026-07-04T00:19:13.718478Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":"2505.15146","doi":"10.48550/arxiv.2505.15146","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xing, Ion Stoica, Tajana Rosing, Haojian Jin, and Hao Zhang","venue":"ArXiv.org","work_id":"10ddd4f7-8f21-466f-b975-f64ea89b9b2e","year":2025},"citing_paper":{"arxiv_id":"2606.24893","last_updated":"2026-05-29T22:40:51Z","snapshot_observed_at":"2026-08-11T16:05:00.071473Z","submitted_at":"2026-05-29T22:40:51Z","title":"AgentOdyssey: Open-Ended Long-Horizon Text Game Generation for Test-Time Continual Learning Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T21:59:25.449570Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2606.24893"},"observation_digest":"sha256:277261e1eaaf7c256e86942da4e47ad8deab047d79703813253838cd3bc92ad3","observation_id":"0c2486de-cfd0-4b84-97a8-f3b4a71f6f00","resolution":{"observed_at":"2026-07-01T19:46:11.447161Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15146","snapshot_observed_at":"2026-08-01T02:58:29.283205Z","title":"arXiv preprint arXiv:2505.15146 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25308","last_updated":"2026-07-28T05:39:01Z","snapshot_observed_at":"2026-08-05T09:28:39.486019Z","submitted_at":"2026-07-28T05:39:01Z","title":"CAST: Game Solvers as Turn-Level Teachers for LLM Agents","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-01T02:58:29.283205Z"},"links":{"cited_paper":"/paper/2505.15146","citing_paper":"/paper/2607.25308"},"observation_digest":"sha256:796ea5191f1050b41e2acd7a2247a7421da7cd39ecedb76a43708abff05abf55","observation_id":"c51221f7-15e7-4980-ac29-d6e0cb128700","resolution":{"observed_at":"2026-08-01T02:58:29.283205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15146/citation-record","integrity":"/paper/2505.15146/integrity","json":"/paper/2505.15146/citation-record.json","paper":"/paper/2505.15146"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-08-09T23:24:21.399948Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-08-07T15:26:55.977387Z","title":"arXiv preprint arXiv:1606.01540 (2016)","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:55.977387Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:8d55398489b8523e71679a19434ae8c69073471b526e14c28fa1b3a56fd174e7","observation_id":"a7e70288-ebc0-4e09-82a6-f0ea6f228c64","resolution":{"observed_at":"2026-08-07T15:26:55.977387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.17032","last_updated":"2025-11-02T13:42:19Z","snapshot_observed_at":"2026-07-06T18:51:06.750136Z","submitted_at":"2024-07-24T06:35:05Z","title":"Gymnasium: A Standard Interface for Reinforcement Learning Environments","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.17032","snapshot_observed_at":"2026-08-07T15:26:56.072057Z","title":"arXiv preprint arXiv:2407.17032 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.072057Z"},"links":{"cited_paper":"/paper/2407.17032","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:40908410aefe86cc72270f479c66e53ff2476312d8d570121fc39bf63762edf4","observation_id":"131c4169-85b2-400a-865b-adda2c05762f","resolution":{"observed_at":"2026-08-07T15:26:56.072057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-08-10T06:52:44.809057Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-08-07T15:26:56.140833Z","title":"arXiv preprint arXiv:2504.20073 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.140833Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:323fa1b78cadab0bc28dabfe297146f780f50b8c7cb43fd4b65d5bcaa38a34dc","observation_id":"ba224b0f-9985-4a21-bb0e-6fa7100a6c5c","resolution":{"observed_at":"2026-08-07T15:26:56.140833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.15478","last_updated":"2025-03-19T17:55:08Z","snapshot_observed_at":"2026-08-08T21:39:18.091217Z","submitted_at":"2025-03-19T17:55:08Z","title":"SWEET-RL: Training Multi-Turn LLM Agents on Collaborative Reasoning Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.15478","snapshot_observed_at":"2026-08-07T15:26:56.255061Z","title":"arXiv preprint arXiv:2503.15478 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.255061Z"},"links":{"cited_paper":"/paper/2503.15478","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3de208a9bb9b73809babb622041ed4e27513f586c59282604992da82d8b6e2ee","observation_id":"14a9e272-3755-4f82-8ebb-4c5d0b603de9","resolution":{"observed_at":"2026-08-07T15:26:56.255061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.363888Z","title":"In Globerson, A., Mackey, L., Belgrave, D., Fan, A., Paquet, U., Tomczak, J., Zhang, C., eds.: Advances in Neural Information Processing Systems","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.363888Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7290f23059c849addd6eb4cdd614925d0605b06aedd61920ae963c03cea37d4b","observation_id":"1d9d2fe6-ad2e-4a8c-ae80-807da734ae0b","resolution":{"observed_at":"2026-08-07T15:26:56.363888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.453904Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.453904Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:1dd46f1bfa8808db32c92e60cfed3f4abc0a4749c51d6fddd335316f2eeea4f1","observation_id":"6d0d7fbc-84b1-43ad-a74d-6b37e571ae55","resolution":{"observed_at":"2026-08-07T15:26:56.453904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.552570Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.552570Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:109eb338015f8f44a6c29b8f6a882526b1939c7b186c96d6fa0a7b099b4ce018","observation_id":"babc8d0b-adff-4d21-9ae3-ee3d6a759645","resolution":{"observed_at":"2026-08-07T15:26:56.552570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08859","last_updated":"2025-02-14T16:40:15Z","snapshot_observed_at":"2026-08-09T16:48:28.493833Z","submitted_at":"2025-02-13T00:18:34Z","title":"EnigmaEval: A Benchmark of Long Multimodal Reasoning Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08859","snapshot_observed_at":"2026-08-07T15:26:56.639836Z","title":"arXiv preprint arXiv:2502.08859 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.639836Z"},"links":{"cited_paper":"/paper/2502.08859","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:06fc206efeb0d05a9b14eb2012be0404d402768a7b45ab7eaf88a11a6118c4cd","observation_id":"57530e12-cda5-41fe-bf7f-6574601f213e","resolution":{"observed_at":"2026-08-07T15:26:56.639836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:56.720112Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.720112Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:41601f1149d4b0470821a97f9f92f20576b8351cfa4b7400996208fafd48bffa","observation_id":"a0ca5e66-1c30-491b-9df9-19c7522ef01d","resolution":{"observed_at":"2026-08-07T15:26:56.720112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-08-10T17:42:21.198695Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-07T15:26:56.816724Z","title":"arXiv preprint arXiv:2411.13543 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.816724Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:76343054c4e5ae3833b79a7f9c7a847424e56f88324b1a0d68da82d86b80a24c","observation_id":"fc84ddec-7530-45b2-aa78-08091fdbc4a6","resolution":{"observed_at":"2026-08-07T15:26:56.816724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06613","last_updated":"2024-07-22T14:32:33Z","snapshot_observed_at":"2026-08-10T10:41:28.705018Z","submitted_at":"2024-06-07T00:28:43Z","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06613","snapshot_observed_at":"2026-08-07T15:26:56.894546Z","title":"arXiv preprint arXiv:2406.06613 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:56.894546Z"},"links":{"cited_paper":"/paper/2406.06613","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:52425c952b553fa19b868ae698d94aa2aa8dd7cce4f00a546e08f4bffb78242b","observation_id":"7e38bd61-6f22-4a25-8709-7c5227b6d909","resolution":{"observed_at":"2026-08-07T15:26:56.894546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01557","last_updated":"2024-03-17T23:23:31Z","snapshot_observed_at":"2026-08-06T16:46:34.632571Z","submitted_at":"2023-10-02T18:52:11Z","title":"SmartPlay: A Benchmark for LLMs as Intelligent Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01557","snapshot_observed_at":"2026-08-07T15:26:57.020933Z","title":"arXiv preprint arXiv:2310.01557 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.020933Z"},"links":{"cited_paper":"/paper/2310.01557","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:1d92aabea27cdb8fe51d22ea6e285a322c11f9a6f8687578138092c6c4ffdebf","observation_id":"feb50f57-f0b3-4573-880d-350a08c99f3b","resolution":{"observed_at":"2026-08-07T15:26:57.020933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.127873Z","title":"IEEE Transactions on Games11(3) (2019) 195–202","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.127873Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dcf09d9aa6fb3a49bb8a80596058fb43411781f85667593522e11db96678cf68","observation_id":"6aa6771e-9efd-40a8-8ccd-f8cd596c8f53","resolution":{"observed_at":"2026-08-07T15:26:57.127873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.204676Z","title":"AI Magazine22(2) (2001) 15–25","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.204676Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:5d00c0e381d3674a7e2706174bf8ad170dcf8a3a4dcaae22863cc136f23abef8","observation_id":"97837f3f-1fdd-4b8d-9c47-82e019956ede","resolution":{"observed_at":"2026-08-07T15:26:57.204676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-10T08:02:13.965614Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-07T15:26:57.341174Z","title":"arXiv preprint arXiv:2412.14171 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.341174Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bba950304ee666d812281b18905b62f7da0a95de93e8506e8556f95c90c96d78","observation_id":"14041c05-4acb-4b6b-bb51-360566b36a2b","resolution":{"observed_at":"2026-08-07T15:26:57.341174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15950","last_updated":"2024-12-02T03:48:43Z","snapshot_observed_at":"2026-08-05T04:23:28.642789Z","submitted_at":"2024-08-28T17:08:56Z","title":"Atari-GPT: Benchmarking Multimodal Large Language Models as Low-Level Policies in Atari Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15950","snapshot_observed_at":"2026-08-07T15:26:57.413647Z","title":"arXiv preprint arXiv:2408.15950 (2024) 11","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.413647Z"},"links":{"cited_paper":"/paper/2408.15950","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ead902b19e1792b5647e1da2ac4a20f36acfeeef3d4d8081bd1b700f85793c5d","observation_id":"2607c706-b810-4f3e-a02e-76c5f1a9e56e","resolution":{"observed_at":"2026-08-07T15:26:57.413647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11381","last_updated":"2024-06-19T16:23:05Z","snapshot_observed_at":"2026-07-06T17:45:59.984749Z","submitted_at":"2024-03-18T00:13:43Z","title":"Can LLM-Augmented autonomous agents cooperate?, An evaluation of their cooperative capabilities through Melting Pot","version":2},"cited_work":{"arxiv_id":"2403.11381","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.11381","snapshot_observed_at":"2026-08-07T15:27:03.880129Z","title":"Can LLM-Augmented autonomous agents cooperate?, An evaluation of their cooperative capabilities through Melting Pot","venue":"cs.AI","work_id":"03b139c8-2038-4fb7-a36c-fca8d92ab30a","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.507753Z"},"links":{"cited_paper":"/paper/2403.11381","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3529092025726611e84ef6606091c32bc4030760956110de7a6ca3333cc81f59","observation_id":"42618b84-232d-4f90-b0d7-8e954d3735c2","resolution":{"observed_at":"2026-08-07T15:27:03.886471Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.604725Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.604725Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:1912417122a40c80125fcd9f28efa87fe5b8fabec4666126357a996ebb982002","observation_id":"0e90353b-7af8-4f73-8b05-213dc2a49d13","resolution":{"observed_at":"2026-08-07T15:26:57.604725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T15:26:57.709482Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.709482Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d0de09aa5affd6839e26c44e549c6681b7404a2cc1003740b25d6b02c666cb0f","observation_id":"3c06fd2b-4bce-456f-ae46-6b275c790620","resolution":{"observed_at":"2026-08-07T15:26:57.709482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.792901Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.792901Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:09eda6388a36b92860f7b1c1db4ea06c1168122db1c896cb8ac590738779bbc3","observation_id":"e2613515-e9ef-4538-b0cb-848ce7429f6a","resolution":{"observed_at":"2026-08-07T15:26:57.792901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.864760Z","title":"In ICAPS","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.864760Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:5fa127b4a5b9151dcc41cb965dc5b83be6b94fff02f5262d6e6ef2d44a7b35f5","observation_id":"f282f996-6b9a-4e1f-8578-ad4a26aa3d85","resolution":{"observed_at":"2026-08-07T15:26:57.864760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:57.936132Z","title":"Applied cognitive psychology31(4) (2017) 438–445","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:57.936132Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:2b52d343bdcb362881c088cdf047d9e30effa5892c774cc099782cb46a9d361e","observation_id":"5ed9dfa1-d833-40fa-af0e-82ebf70154f0","resolution":{"observed_at":"2026-08-07T15:26:57.936132Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.046424Z","title":"In International Computing and Combinatorics Conference (COCOON)","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.046424Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:33e58bf522cf27e70f06706f5c3fee0b4f25fc0007aa419daf5ebc98daec8b27","observation_id":"a3791a52-c555-4ec5-bb2f-a3ea2101db68","resolution":{"observed_at":"2026-08-07T15:26:58.046424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.163042Z","title":null,"venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.163042Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9d54e5bf35a81d516a4d4dd53d083727f6220e82731838533d4f5449d96b8092","observation_id":"a661cdd9-5b21-433f-b165-7d7818203605","resolution":{"observed_at":"2026-08-07T15:26:58.163042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.220901Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.220901Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ed150d70f9cd5d43c1901c9353a3daba723f5a1fc8dc76e9e596d1b29847c467","observation_id":"458995a7-b87c-48cf-892d-495c8af4cd2b","resolution":{"observed_at":"2026-08-07T15:26:58.220901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.280311Z","title":"arXiv (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.280311Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:c38943cc6277e3ab7c04e8f3bf2ec690de6c50ddba1a5b8726f9b6661d5b91c6","observation_id":"68d50081-a23e-4b11-ad68-4a9665b646da","resolution":{"observed_at":"2026-08-07T15:26:58.280311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.353668Z","title":"arXiv (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.353668Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4a9116897b55ab077978ea9c6ae1a54a71a2dd5844c27992207f64f18d17537c","observation_id":"eadfffa4-9db8-4542-9f36-cc139a0416a3","resolution":{"observed_at":"2026-08-07T15:26:58.353668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T15:26:58.427826Z","title":"arXiv preprint arXiv:2501.12948 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.427826Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bc19de4a21d6971e46e8ce357b8e711522c71a0113fd47eb220fed35efffc235","observation_id":"b9973ff4-17fc-4c65-a76e-ebac676753b3","resolution":{"observed_at":"2026-08-07T15:26:58.427826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.487599Z","title":"Computational Geometry 13(4) (1999) 215–228","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.487599Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f5faed9ee41ff880e70bc4644913ad29781cd74fa1176514b4fbb9959f9c803d","observation_id":"24d32840-27ff-4a3a-95c4-4e9206bb84f4","resolution":{"observed_at":"2026-08-07T15:26:58.487599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1403.5484","last_updated":"2014-03-21T15:03:33Z","snapshot_observed_at":"2026-07-06T03:39:07.358584Z","submitted_at":"2014-03-21T15:03:33Z","title":"A Simple Family of Analytical Trumpet Slices of the Schwarzschild Spacetime","version":1},"cited_work":{"arxiv_id":"1403.5484","doi":null,"metadata_source":"pith","pith_arxiv_id":"1403.5484","snapshot_observed_at":"2026-08-07T15:27:03.815957Z","title":"A Simple Family of Analytical Trumpet Slices of the Schwarzschild Spacetime","venue":"gr-qc","work_id":"f4d774ae-ac75-4564-b1a9-c2dddd2e2384","year":2014},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.543471Z"},"links":{"cited_paper":"/paper/1403.5484","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:04d2872f3cabc9a59106f617aae41b8d618bb3648f96336e5dda81b8aee8d372","observation_id":"315386b0-a926-4c90-b7d8-da2b71f4b8b2","resolution":{"observed_at":"2026-08-07T15:27:03.824241Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15938","last_updated":"2024-05-31T17:49:03Z","snapshot_observed_at":"2026-08-10T09:23:20.862829Z","submitted_at":"2024-02-24T23:54:41Z","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15938","snapshot_observed_at":"2026-08-07T15:26:58.602648Z","title":"arXiv preprint arXiv:2402.15938 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.602648Z"},"links":{"cited_paper":"/paper/2402.15938","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a93faf08619fe6354f4c57d4a5d4cf4352cfda14d9977415f6903922dca0b8f2","observation_id":"df103d95-f26f-4379-a55b-5e402af91911","resolution":{"observed_at":"2026-08-07T15:26:58.602648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1802.08232","last_updated":"2019-07-16T17:05:32Z","snapshot_observed_at":"2026-08-06T23:45:16.145644Z","submitted_at":"2018-02-22T18:42:41Z","title":"The Secret Sharer: Evaluating and Testing Unintended Memorization in Neural Networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1802.08232","snapshot_observed_at":"2026-08-07T15:26:58.673836Z","title":"arXiv preprint arXiv:1802.08232 (2018)","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.673836Z"},"links":{"cited_paper":"/paper/1802.08232","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bcb0a0e7281f9b5ebc86e535b36d397bed7f36ad3abd6ed3bb061f20f4abb992","observation_id":"b6f11591-6ffc-4ad2-a2a3-a1758280f791","resolution":{"observed_at":"2026-08-07T15:26:58.673836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.734689Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.734689Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:08d495f9229aa89824ebe010de421a40d5005fd7225642ced1ac64c9bf286071","observation_id":"779eca80-188b-4c50-852e-178d48195594","resolution":{"observed_at":"2026-08-07T15:26:58.734689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.819840Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.819840Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:03b5aab34c5745fbcdb3ac15655f1eb276b0c4a3ba9f9ae9600b6ebc70a65c90","observation_id":"c90c83f6-52ab-4545-84d1-3304d47f63c3","resolution":{"observed_at":"2026-08-07T15:26:58.819840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03186","last_updated":"2024-07-02T17:23:13Z","snapshot_observed_at":"2026-08-09T09:54:10.383680Z","submitted_at":"2024-03-05T18:22:29Z","title":"Cradle: Empowering Foundation Agents Towards General Computer Control","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03186","snapshot_observed_at":"2026-08-07T15:26:58.872288Z","title":"arXiv preprint arXiv:2403.03186 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.872288Z"},"links":{"cited_paper":"/paper/2403.03186","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f6eb7ea7f4df34af3a4fc909a0ae02678d8bd6e87b8be004207063fe8e24fbaa","observation_id":"5a38610f-7acc-42ac-a56d-c9ab3c9d4908","resolution":{"observed_at":"2026-08-07T15:26:58.872288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:58.941657Z","title":"In The Twelfth International Conference on Learning Representations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:58.941657Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4f0853c7708340af1d4e7469d015c26ad71617b0c8285a6e382f81df1f928e51","observation_id":"091a4179-5910-422c-8688-a32d3cb84c84","resolution":{"observed_at":"2026-08-07T15:26:58.941657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.038784Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.038784Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:bd7c9fdb9aeab9861f38a64cb55fee8581f2536f56af010890df18417debf516","observation_id":"e14d8a20-afe8-440c-b5d0-b4b75a82f599","resolution":{"observed_at":"2026-08-07T15:26:59.038784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-10T12:35:09.020030Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T15:26:59.115491Z","title":"arXiv preprint arXiv:2009.03300 (2021)","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.115491Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9f583b01c92fc865e0514a1da30fabfadde0c9075efc47d8543564b61c869ff2","observation_id":"1f4eaf26-1294-4981-812f-a1f1136a981e","resolution":{"observed_at":"2026-08-07T15:26:59.115491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-07T15:26:59.189173Z","title":"arXiv preprint arXiv:2501.14249 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.189173Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f6a0e5f2e2d08d185f9fbaac359b3b43e44e2a6ea7db9fa43c9cadad3a0fe293","observation_id":"78264078-8bb5-473a-a8db-9465570d5a75","resolution":{"observed_at":"2026-08-07T15:26:59.189173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.256804Z","title":"https://scale.com/leaderboard/ humanitys_last_examAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.256804Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:669a1cf859d0ffb20d2d896c14ce3a19fc3fda1e75dc51246b6cdee46a5a5ca9","observation_id":"5215fa44-fcc9-4428-ac26-30c80fcaa06c","resolution":{"observed_at":"2026-08-07T15:26:59.256804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.333175Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.333175Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cf8bfb941000c141ec95d3ff734d7a35b4ecb8b5df4373b7aa950d8ea5b54021","observation_id":"fa21a2a7-b374-4ee5-8416-5e1e6fa90698","resolution":{"observed_at":"2026-08-07T15:26:59.333175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-10T12:02:35.919497Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-07T15:26:59.405000Z","title":"arXiv preprint arXiv:2311.12022 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.405000Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cfbb5df192a224ddbabc18f474531bb2fef76ada747634ce8f05cf84d2db815e","observation_id":"05197398-9457-479b-8f9b-c55b05938893","resolution":{"observed_at":"2026-08-07T15:26:59.405000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.464341Z","title":"https://www.vals.ai/benchmarks/ gpqa-05-09-2025Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.464341Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dbc29304e1897f8b89241189109f2903b2e946c3dc9d2e08324c1f7d566d6ef1","observation_id":"18401cb3-ae82-436f-a734-1c1b8cf03368","resolution":{"observed_at":"2026-08-07T15:26:59.464341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16074","last_updated":"2025-05-18T14:13:34Z","snapshot_observed_at":"2026-08-07T16:00:09.967849Z","submitted_at":"2025-04-22T17:53:29Z","title":"PHYBench: Holistic Evaluation of Physical Perception and Reasoning in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.16074","snapshot_observed_at":"2026-08-07T15:26:59.524310Z","title":"arXiv preprint arXiv:2504.16074 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.524310Z"},"links":{"cited_paper":"/paper/2504.16074","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:3772f4b3a4a9986ddeb7243fe9ed4a37560f9b8d66db0f6466f74642c820e27a","observation_id":"c0b9f034-2e97-42c1-bb1a-d5525b541192","resolution":{"observed_at":"2026-08-07T15:26:59.524310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05444","last_updated":"2025-01-09T18:55:52Z","snapshot_observed_at":"2026-08-10T21:11:43.711209Z","submitted_at":"2025-01-09T18:55:52Z","title":"Can MLLMs Reason in Multimodality? EMMA: An Enhanced MultiModal ReAsoning Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05444","snapshot_observed_at":"2026-08-07T15:26:59.566161Z","title":"arXiv preprint arXiv:2501.05444 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.566161Z"},"links":{"cited_paper":"/paper/2501.05444","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:6427d3a656dc0f0cc3635c9c04321c2143246224e386485c6e1427fdf42d9596","observation_id":"8cafd74e-8f58-4fde-a158-02ea21ec09c9","resolution":{"observed_at":"2026-08-07T15:26:59.566161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.676392Z","title":"https://www.vals.ai/benchmarks/ math500-05-09-2025Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.676392Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:56c2e190c5f949594226d094febe467f2ad1f4e7a624b16cdb4238f0ef4f16e0","observation_id":"cce49f77-c6ae-4fb2-aab9-75b4ec1248d6","resolution":{"observed_at":"2026-08-07T15:26:59.676392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.788838Z","title":"arXiv preprint arXiv:2410.03131 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.788838Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:0a2426bc277e6c19d73b948a1ab98a8b4528510b3f7795efded1d56a4556cf80","observation_id":"90b9f4e3-156d-4664-aee1-57fb91077fdf","resolution":{"observed_at":"2026-08-07T15:26:59.788838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:26:59.975697Z","title":"https://www.vals.ai/benchmarks/ aime-2025-05-09Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:26:59.975697Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:6c15f89eccd8d05fa8ee6550dde4423af133bfee6bd30e568efcb5f6e77498e1","observation_id":"fd08dd1e-6fbf-48a1-98b3-78fd6a87369b","resolution":{"observed_at":"2026-08-07T15:26:59.975697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-07T15:27:00.152304Z","title":"arXiv preprint arXiv:2406.19314 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.152304Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:baf12e01009654371a0132b0b3104fdeacca4e2750c7be0cc60118d5f033c244","observation_id":"b2b1ae1a-b149-46a6-8cf8-7810df23e33b","resolution":{"observed_at":"2026-08-07T15:27:00.152304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.352797Z","title":"https://livebench.ai/#/?Coding=a& Mathematics=a&Data+Analysis=a&Language=a&IF=aAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.352797Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:03bb4fa09bc4d0719aeb24c5bc83ad92aaf5d4bd471b3e85a1d5a588878d7ef5","observation_id":"57ad3dd4-a046-4304-b642-c79511ba15da","resolution":{"observed_at":"2026-08-07T15:27:00.352797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15877","last_updated":"2025-04-01T08:36:44Z","snapshot_observed_at":"2026-07-31T19:00:59.311189Z","submitted_at":"2024-06-22T15:52:04Z","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15877","snapshot_observed_at":"2026-08-07T15:27:00.559765Z","title":"arXiv preprint arXiv:2406.15877 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.559765Z"},"links":{"cited_paper":"/paper/2406.15877","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f2c13d3a9ef8f6f5344608b9acc34ccb51e59ae32554f99a93fb3939469d8294","observation_id":"57249563-1ef7-4de2-a8ef-873d95490edd","resolution":{"observed_at":"2026-08-07T15:27:00.559765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.736996Z","title":"https://aider.chat/docs/leaderboards/ Ac- cessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.736996Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:e5d165dc233c29cff9f3cc1acffcc774a12aab9f7ea73f88843b2a308a5e3dfa","observation_id":"417548ee-d2f4-4b93-a41b-3768f88ef4a6","resolution":{"observed_at":"2026-08-07T15:27:00.736996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.875924Z","title":"https://bigcode-bench.github.io/ Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.875924Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:cbe7b3feb20239be26e0c1f705f84954ad02ca847b5d6716f7bab92a4748b4f2","observation_id":"f74dbb8d-05cd-47d5-805c-429e6fb2df25","resolution":{"observed_at":"2026-08-07T15:27:00.875924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.977620Z","title":"https://scale","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.977620Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:4e14364410dd77d023e4ee1d0f3c10c7443c2cd4852f609c3924e41b37f2b20d","observation_id":"5f981890-9fcb-428b-ba47-a854b244afe3","resolution":{"observed_at":"2026-08-07T15:27:00.977620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:00.981839Z","title":"https://lmarena.ai/leaderboard Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.981839Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:6aa8da61d0756ed3ae419086be9bf851edd930903621f5fb2cbfd0f8d3a24ab9","observation_id":"1b3e83b5-e1af-4700-83e5-b546bd993e2a","resolution":{"observed_at":"2026-08-07T15:27:00.981839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16502","last_updated":"2024-06-13T15:02:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-27T17:33:21Z","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16502","snapshot_observed_at":"2026-08-07T15:27:00.986636Z","title":"arXiv preprint arXiv:2311.16502 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:00.986636Z"},"links":{"cited_paper":"/paper/2311.16502","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:403ec9d28c05de0b175c851f432e696eabbf12ad95872756a08be5c096b24953","observation_id":"c8b4e03f-6aa8-40d0-9580-048940736906","resolution":{"observed_at":"2026-08-07T15:27:00.986636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.031667Z","title":"https://www.vals.ai/benchmarks/ mmmu-05-09-2025Accessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.031667Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f98d279ab7088031c2e1c2aa9894f1770d022df45737640557fd91dfeffd74f2","observation_id":"46b6f099-93db-472a-bbb9-5991bf80b0b2","resolution":{"observed_at":"2026-08-07T15:27:01.031667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17399","last_updated":"2025-03-06T04:41:56Z","snapshot_observed_at":"2026-08-10T04:40:42.053539Z","submitted_at":"2025-01-29T03:29:24Z","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17399","snapshot_observed_at":"2026-08-07T15:27:01.110876Z","title":"arXiv preprint arXiv:2501.17399 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.110876Z"},"links":{"cited_paper":"/paper/2501.17399","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:fe9bcd50bb1a572712f0e4e90a1e9c752d4cc4bb66946a529387886835c6eef1","observation_id":"8158aad3-a061-4172-a5e1-392d3e4fa3a4","resolution":{"observed_at":"2026-08-07T15:27:01.110876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.208322Z","title":"https://scale.com/leaderboard/ multichallengeAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.208322Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:8cab7150651aa32bbaaaf1c8ee8c7e487ff2542bc0097bf3f4cd4f2fec2d9155","observation_id":"3e400887-44cc-4397-99cf-e6dcb36ebaa9","resolution":{"observed_at":"2026-08-07T15:27:01.208322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.299976Z","title":"https://scale.com/leaderboard/ enigma_evalAccessed: 2025-05-14","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.299976Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:d2068b1620c0e0c6b4b0c00f2017bd288dae94d878d97c451aaf91fd1e152cb5","observation_id":"e543a205-cfd9-4a73-8081-570c10c774b0","resolution":{"observed_at":"2026-08-07T15:27:01.299976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.362623Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.362623Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:e2d3909856424a270c360968ba117b29429911b98a022d3eacdd9b40399a9cf6","observation_id":"d1b43e71-9dfc-4115-b0db-dde5f55ad797","resolution":{"observed_at":"2026-08-07T15:27:01.362623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:01.442781Z","title":"https://github.com/mpSchrader/gym-sokoban (2018)","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.442781Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:5434ccaf328a088ab438a7731bea67371731ba2ff6cb0d8bd3cf79474faaebcc","observation_id":"9b944751-7bca-4f85-a541-36dd67dd2b0b","resolution":{"observed_at":"2026-08-07T15:27:01.442781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.620840Z","title":"https://github.com/jaybutera/ tetrisRL(2023) GitHub repository","venue":null,"work_id":"12cde6ce-d822-4595-840c-80ee4e52b193","year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.518032Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:72c99b43691859864bc2820bb684f0e00f26eb8bc25e04d4a6ed4767a3c5f364","observation_id":"942416c6-c81b-4b08-a263-8b1b5a540611","resolution":{"observed_at":"2026-08-07T15:27:04.626893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T15:27:01.523920Z","title":"5 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.523920Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:69db724d9cb852c88cb5712bc7afee7ed356224d568e433b560a43589c0c0813","observation_id":"086da939-2779-4dbc-9d8a-e528b44e84a9","resolution":{"observed_at":"2026-08-07T15:27:01.523920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.604164Z","title":"Communications of the ACM 38(3) (1995) 58–68","venue":null,"work_id":"a1177d46-09d8-4309-813b-77c837d6615f","year":1995},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.528538Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f628c045b4f341be45dbeb349321badc21054f99e98a5fe8c5e9175887cece3f","observation_id":"cf891336-6b2c-40d8-abd9-19785f67c849","resolution":{"observed_at":"2026-08-07T15:27:04.609837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.587428Z","title":"nature550(7676) (2017) 354–359","venue":null,"work_id":"c8f84a29-defd-4e52-b390-06db19078ba3","year":2017},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.544702Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:09f2ab04e238cac3b54db2b676a1a50d720effed8dabdb1c4e485fc37b707bc2","observation_id":"b0d4f444-27e6-4395-a8e0-5ae73574d782","resolution":{"observed_at":"2026-08-07T15:27:04.592570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.571251Z","title":"In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track","venue":null,"work_id":"d1c0e405-701f-4faf-95c9-dbf239657d4f","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.617277Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7b58e9ff8dc28162e211f59b83354df87cbbb1201df008785508b5e501434d5b","observation_id":"04da8a84-7fdb-4484-9b9a-ef622d288117","resolution":{"observed_at":"2026-08-07T15:27:04.576974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09617","last_updated":"2025-03-06T20:13:02Z","snapshot_observed_at":"2026-08-07T17:23:37.144054Z","submitted_at":"2025-03-06T20:13:02Z","title":"Factorio Learning Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09617","snapshot_observed_at":"2026-08-07T15:27:01.689046Z","title":"arXiv preprint arXiv:2503.09617 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.689046Z"},"links":{"cited_paper":"/paper/2503.09617","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a46c7a5ced47df5ad4fd115e93720fdaa7b006fc650f4814f13da9b42c097f47","observation_id":"f9065042-7769-4785-a445-20b67486def2","resolution":{"observed_at":"2026-08-07T15:27:01.689046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.555948Z","title":"In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track","venue":null,"work_id":"ade258e3-23d9-4187-9990-05ce319dc3f8","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.766369Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:94687f98d27052bc43e728f66960dfb6daa785d8c96457cc886054caf9b307ec","observation_id":"9a6fa030-734c-47ee-ad2d-e8c8af2764d1","resolution":{"observed_at":"2026-08-07T15:27:04.560892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18431","last_updated":"2025-02-25T18:26:48Z","snapshot_observed_at":"2026-08-07T17:49:22.453162Z","submitted_at":"2025-02-25T18:26:48Z","title":"TextGames: Learning to Self-Play Text-Based Puzzle Games via Language Model Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18431","snapshot_observed_at":"2026-08-07T15:27:01.826741Z","title":"arXiv preprint arXiv:2502.18431 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.826741Z"},"links":{"cited_paper":"/paper/2502.18431","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ed2ce2284a5104261a25de86e9c82a1923d84f37b662b69d20188bc4ca8d924d","observation_id":"16b11839-c133-4a36-bcd0-05ae76aa1262","resolution":{"observed_at":"2026-08-07T15:27:01.826741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10032","last_updated":"2023-08-19T14:33:40Z","snapshot_observed_at":"2026-07-06T16:08:00.845026Z","submitted_at":"2023-08-19T14:33:40Z","title":"GameEval: Evaluating LLMs on Conversational Games","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10032","snapshot_observed_at":"2026-08-07T15:27:01.892670Z","title":"arXiv preprint arXiv:2308.10032 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.892670Z"},"links":{"cited_paper":"/paper/2308.10032","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dcb9da43a41633c776ec4d6035ed0bac71e52a70dccc5e259eb24f586a77905f","observation_id":"ef4dd5ca-d0f2-4830-a96b-26312327dea1","resolution":{"observed_at":"2026-08-07T15:27:01.892670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06394","last_updated":"2025-02-15T22:03:16Z","snapshot_observed_at":"2026-08-06T14:14:54.915113Z","submitted_at":"2024-12-09T11:22:59Z","title":"GameArena: Evaluating LLM Reasoning through Live Computer Games","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06394","snapshot_observed_at":"2026-08-07T15:27:01.953231Z","title":"arXiv preprint arXiv:2412.06394 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:01.953231Z"},"links":{"cited_paper":"/paper/2412.06394","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:0fe85bed10b7081f2461e833644a90af973535b2faef2bc18bd63a8f8ada6efb","observation_id":"ff01265b-d150-4cf4-bf94-37cd01ea52a9","resolution":{"observed_at":"2026-08-07T15:27:01.953231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-07T15:27:02.019566Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.019566Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:70da901182c5bc4d8051172a1a99a312ea690ce37a05111b00096228cc6214a6","observation_id":"6a8c9fc0-bf4f-4d18-a4ec-336252fdc380","resolution":{"observed_at":"2026-08-07T15:27:02.019566Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-06T12:47:01.809185Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-07T15:27:02.059918Z","title":"arXiv preprint arXiv:2307.13854 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.059918Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:35fb0619568456da3326587b0a42f6e9be3604a1b44f9ea7e9539bb932e67d94","observation_id":"93c0d409-ab27-49cb-8248-b6a7e5e39c44","resolution":{"observed_at":"2026-08-07T15:27:02.059918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13919","last_updated":"2024-06-06T18:37:34Z","snapshot_observed_at":"2026-08-09T21:47:22.232349Z","submitted_at":"2024-01-25T03:33:18Z","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13919","snapshot_observed_at":"2026-08-07T15:27:02.065525Z","title":"arXiv preprint arXiv:2401.13919 (2024) 14","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.065525Z"},"links":{"cited_paper":"/paper/2401.13919","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:699d0a66b5506f144439ecd8b2f6890132519787e404199edfe48740b388b251","observation_id":"71003540-eb94-4f0a-8628-80fdb703865a","resolution":{"observed_at":"2026-08-07T15:27:02.065525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18901","last_updated":"2024-07-26T17:55:45Z","snapshot_observed_at":"2026-08-11T15:14:53.237571Z","submitted_at":"2024-07-26T17:55:45Z","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18901","snapshot_observed_at":"2026-08-07T15:27:02.082501Z","title":"arXiv preprint arXiv:2407.18901 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.082501Z"},"links":{"cited_paper":"/paper/2407.18901","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:eb6cc462bf67f05ffc7b9d3c412834be42644e8c089cbaa7b8d06502b899ad33","observation_id":"fbbce248-f0d4-44ce-9218-ff4a600c30a9","resolution":{"observed_at":"2026-08-07T15:27:02.082501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.540766Z","title":"Advances in Neural Information Processing Systems37(2024) 52040–52094","venue":null,"work_id":"c5d0e6f1-ff42-4806-90dd-2443a043a1e5","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.131440Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9dd974bf3cc6cbfdd3d6a1d3410cd485a3c012c4193f58961eeb52594ae7563f","observation_id":"55c10a9a-0b26-4173-92d4-352510af82cc","resolution":{"observed_at":"2026-08-07T15:27:04.545710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.524038Z","title":null,"venue":null,"work_id":"0ec30dc5-50e6-439e-90d5-82b9dfa75fe5","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.208401Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:b3ccd664a122cc53fa3d1579a7cbbd16c5ffa5b2002f87833cdd7b1ec295390c","observation_id":"23603768-7adc-4214-82f5-134ebcf65371","resolution":{"observed_at":"2026-08-07T15:27:04.528964Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.508562Z","title":"In The Twelfth International Conference on Learning Representations","venue":null,"work_id":"3f209b8d-ebae-4ea0-b234-eea563911356","year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.293383Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ec85ef0cb343d0d969a5990c1de4c697a02a98920d15f5efa42e3436732a721a","observation_id":"ab6ef914-7e02-48fc-ac4c-3a0923f26339","resolution":{"observed_at":"2026-08-07T15:27:04.513351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.490748Z","title":"Advances in neural information processing systems37(2024) 110935–110971","venue":null,"work_id":"e47933dd-7f2d-4b60-aec3-703d80ea4a66","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.328494Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:8a9b5aa85faffe6b3d057051c633284d76aa6aaa57ccba64a07f3e69f3640867","observation_id":"ac079a74-203a-44ca-a45c-7e4e7b81e5a3","resolution":{"observed_at":"2026-08-07T15:27:04.497010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T15:27:02.348919Z","title":"arXiv preprint arXiv:1707.06347 (2017)","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.348919Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f2efb88de221fec016c9a3e52b971505a5b636f2c2b48940b7f8be63318bc5ef","observation_id":"dd33cdcf-25ab-4ac1-8e0a-508f1a56a9f0","resolution":{"observed_at":"2026-08-07T15:27:02.348919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.473365Z","title":"Science10(3) (1995) 237–304","venue":null,"work_id":"4e8a52d5-7eb0-4c39-95ce-aa00449b0888","year":1995},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.397270Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:60cfbb4fa235d718d1aea80d7009f3a63396f69b965e90f2f1e8ef0dfb214714","observation_id":"f23d2550-8302-4b52-9bb2-70b39c762a73","resolution":{"observed_at":"2026-08-07T15:27:04.478783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-07T15:27:02.445271Z","title":"arXiv preprint arXiv:2503.14476 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.445271Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a00e33ad0647adc36be6e9209b471b93361351d32b75bdc4b715375de8db8e0c","observation_id":"d17c69fb-0447-4821-94ee-8b7e972aff5c","resolution":{"observed_at":"2026-08-07T15:27:02.445271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.456650Z","title":"Advances in Neural Information Processing Systems36(2023) 38975–38987","venue":null,"work_id":"5366aa45-3804-4012-8e2c-9620f6996297","year":2023},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.509094Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7b14d7377e690642db48f075506f5118af51a450232a6d2b6a434c5bbcf92027","observation_id":"bc3d9d72-699d-4e55-a952-6d4994acc0ff","resolution":{"observed_at":"2026-08-07T15:27:04.461822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T15:27:02.575808Z","title":"arXiv preprint arXiv:2110.14168 (2021)","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.575808Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:7637cab593aebfbbc2462d8b065f4156908277c62970c8e1922bf7ad332ebf1b","observation_id":"87ddd352-621e-4aab-a607-a901a790b5ca","resolution":{"observed_at":"2026-08-07T15:27:02.575808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.438486Z","title":"Advances in Neural Information Processing Systems36(2024)","venue":null,"work_id":"adca9e0a-b4bb-47f0-aab7-090f16bd7a06","year":2024},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.602474Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f39a93c428da959066ca54111276dbe24256ce93aae4588377e4da60ca2865bd","observation_id":"102da256-cfb9-408f-8b4a-80014645a6b9","resolution":{"observed_at":"2026-08-07T15:27:04.444914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.421210Z","title":"Advances in Neural Information Processing Systems 35(2022) 20744–20757","venue":null,"work_id":"79f1fb76-74a3-43fb-bf08-fb8278a01cc0","year":2022},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.606960Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:ab984246067616407b0513821ce2dfefe6673c1bafa9c8782757eefb99dcc5c9","observation_id":"25bdbe78-156d-4464-be03-03384ac850ac","resolution":{"observed_at":"2026-08-07T15:27:04.426651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.404265Z","title":"Biometrika30(1/2) (1938) 81–93","venue":null,"work_id":"c6f01a98-5400-42dc-8307-51d09f41cc9c","year":1938},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.679885Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:92b2dd78d2b696cd8df8c2d81783ea6a9d7f778bb9e1d155bacbf41b7f0959e5","observation_id":"82404a63-dd5e-4dc3-bfea-47ec0b60f14c","resolution":{"observed_at":"2026-08-07T15:27:04.409726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.387745Z","title":"In Proceedings of the 19th international conference on World wide web, ACM (2010) 577–586","venue":null,"work_id":"9cade593-f848-4223-a218-637af6e5cdb7","year":2010},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.740913Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:c4f5b9ef2c1d871af121e9824c336cfd5e2886606c361ef0da31755178ba7334","observation_id":"49d0ee1f-ec40-4b19-b216-87c56e6c6c61","resolution":{"observed_at":"2026-08-07T15:27:04.392662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.371875Z","title":"Educational Researcher5(10) (1976) 3–8","venue":null,"work_id":"40f93839-b076-46bb-b1da-12612a857309","year":1976},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.809560Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:f29909d643f4cbb2579baf00fdf56c964a6c84cbb0d06e6bd1f5f56834410b38","observation_id":"4f03dc3a-ea7e-47b1-8691-d7281719da5a","resolution":{"observed_at":"2026-08-07T15:27:04.376853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.355536Z","title":"Block 1 is on top of block 3, block 3 is on top of block 2, and block 2 is on the table","venue":null,"work_id":"2ece6ea1-1eb5-4a85-b840-d35073e9d889","year":1908},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.852880Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:83bf68a8bd03ddbfa67b8c435c4f5304f7a05a70b50df518b4f2e3ef757d1b55","observation_id":"fb942280-1434-4e92-8e42-927df2631386","resolution":{"observed_at":"2026-08-07T15:27:04.361377Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:02.911988Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.911988Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:33e2524a24d0d8432a89b812d1121b07c8771334a5861e62866d1cd41af20b84","observation_id":"616a293f-add6-4048-8061-6e2fee8b6378","resolution":{"observed_at":"2026-08-07T15:27:02.911988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:02.950794Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:02.950794Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:dc0ab5f04bb99b30870fc86c64b7f0b7998f3c3db4a205800635e96db4ca58b2","observation_id":"9dda3d17-5a70-466d-8d56-55981e5b0ef1","resolution":{"observed_at":"2026-08-07T15:27:02.950794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.032310Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.032310Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a97e4091b35f60649388612643d58a8647dd1b07e6a50d2e8465ccff9650261f","observation_id":"e1830c7f-0d00-443b-9b70-32409578e2b0","resolution":{"observed_at":"2026-08-07T15:27:03.032310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.119984Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.119984Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:79bda64b8e28952b9e0d0159611427a7f4dcd9acdd9f0c4c779b811128c64477","observation_id":"56d4f2d6-0d9a-4a9e-8129-55c3b56a358f","resolution":{"observed_at":"2026-08-07T15:27:03.119984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:04.298615Z","title":"up\", \"down","venue":null,"work_id":"2730b75d-2248-4788-9fcf-70c81694822d","year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.163759Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:0a140508fca2aaa405dae23654ebc50ec45388aa5252d5aff7e830cc4c6317f4","observation_id":"fff924be-933e-4c74-bb34-f0059920b182","resolution":{"observed_at":"2026-08-07T15:27:04.304373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.171436Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.171436Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:35b7935342922bbc4c89a6d92d13408055d53b53994abb227784bc4408398452","observation_id":"ce3e4880-c2cc-45a0-ae20-063acc143f72","resolution":{"observed_at":"2026-08-07T15:27:03.171436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.177177Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.177177Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:68337c19f8c763f6a713b0c7ce2d68df932a2544611ba385c812cd06db22538f","observation_id":"f08e1eca-d190-4a28-b8c3-50d5a956d03c","resolution":{"observed_at":"2026-08-07T15:27:03.177177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.182638Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.182638Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:a485400d1322831cb4dcba6729b9e8d068e3ba7ef65ca77d4729b136af7dc739","observation_id":"1025f4c9-f81b-4b4d-b541-07f78456c050","resolution":{"observed_at":"2026-08-07T15:27:03.182638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:03.187728Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?","version":2},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:03.187728Z"},"links":{"citing_paper":"/paper/2505.15146"},"observation_digest":"sha256:9fb7148afcc9cdde0b5ae2b5f37ff2c47e6fa347fe5f1f92366ec1a1ece9c828","observation_id":"c58fb195-9556-4c8d-ae4b-178daf346b75","resolution":{"observed_at":"2026-08-07T15:27:03.187728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.15146","last_updated":"2025-06-03T09:53:37Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-10T08:39:37.621137Z","submitted_at":"2025-05-21T06:02:55Z","title":"lmgame-Bench: How Good are LLMs at Playing Games?"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":81,"verified_exact":0,"verified_fuzzy":16},"total_outbound_references":115},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 100 of 115 outbound references and 22 inbound Pith citation observations for arXiv:2505.15146."}