{"as_of":"2026-08-04T18:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a8718fa3da8e9b47131203d6ede50714d738075735058275fda677ec645f2d8f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T05:32:20.374639Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2506.03610","last_updated":"2026-04-14T22:54:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-04T06:40:33Z","title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-19T12:01:42.681135Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2506.03610"},"observation_digest":"sha256:f102c92eb31caed465edec683a021a1fc95c7a1eea2e593fec4f934e350f0113","observation_id":"b0f44da1-aa5d-489c-9942-f79c07a69933","resolution":{"observed_at":"2026-05-19T12:02:16.708806Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2511.00739","last_updated":"2026-04-16T18:23:56Z","snapshot_observed_at":"2026-08-03T23:57:10.965977Z","submitted_at":"2025-11-01T23:46:44Z","title":"Towards Understanding, Analyzing, and Optimizing Agentic AI Execution: A CPU-Centric Perspective","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-18T00:55:14.213746Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2511.00739"},"observation_digest":"sha256:b168d69f7f2c8c96f9623850fe963010f1278c279110840f119d11b945d44da0","observation_id":"4b2f3b3e-49b7-459a-ad5b-4048162b3566","resolution":{"observed_at":"2026-05-18T00:55:33.414341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-03T00:13:27.793264Z","title":"Balrog: Benchmarking agentic llm and vlm reasoning on games.arXiv preprint arXiv:2411.13543,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.793264Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:37bfad88fd69ab3b1841f521922c5169f072ceb6d0ee4e81fb3ae968a8a6425a","observation_id":"97c2fc39-6466-4cf3-a527-32042fbe3344","resolution":{"observed_at":"2026-08-03T00:13:27.793264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-02T22:26:37.148467Z","title":"Balrog: Benchmarking agentic llm and vlm reasoning on games","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.16902","last_updated":"2026-05-30T16:45:37Z","snapshot_observed_at":"2026-08-02T22:26:33.125867Z","submitted_at":"2026-02-18T21:33:59Z","title":"LLM-WikiRace Benchmark: How Far Can LLMs Plan over Real-World Knowledge Graphs?","version":4},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T22:26:37.148467Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2602.16902"},"observation_digest":"sha256:c759b5db524995c3f03ae5a0f74a5cdeae296dbf84c87723a5b6e49bacab3b8d","observation_id":"947c1b49-e8f2-402d-b200-91c5492f9ae9","resolution":{"observed_at":"2026-08-02T22:26:37.148467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2604.08340","last_updated":"2026-08-03T15:01:45Z","snapshot_observed_at":"2026-08-04T18:34:26.928600Z","submitted_at":"2026-04-09T15:12:36Z","title":"Mastering PokeGym: Graph-Guided Multimodal Evolution at Test Time","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T17:59:48.877783Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2604.08340"},"observation_digest":"sha256:b9c540160c5b4aad8ea77071dae1113b09a7f9dd124fd7a9f8a9b25c46742fcb","observation_id":"be4c6fd2-0156-40a1-9993-3bf9027a06ee","resolution":{"observed_at":"2026-05-11T05:41:00.454858Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-04T05:32:20.374639Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.08340","last_updated":"2026-08-03T15:01:45Z","snapshot_observed_at":"2026-08-04T18:34:26.928600Z","submitted_at":"2026-04-09T15:12:36Z","title":"Mastering PokeGym: Graph-Guided Multimodal Evolution at Test Time","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T05:32:20.374639Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2604.08340"},"observation_digest":"sha256:841e3b2c489047d2228732d2df9f1675d2765a8dca8b2c94520d207225eceddf","observation_id":"b3475f7c-cfe5-4a63-b749-b21218b45b5d","resolution":{"observed_at":"2026-08-04T05:32:20.374639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2604.20987","last_updated":"2026-04-22T18:17:17Z","snapshot_observed_at":"2026-07-06T23:07:36.996629Z","submitted_at":"2026-04-22T18:17:17Z","title":"Co-Evolving LLM Decision and Skill Bank Agents for Long-Horizon Tasks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T00:14:07.017420Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2604.20987"},"observation_digest":"sha256:d7f6f7bff6d66e0d6fe43bd1b4ed621253f68ba8f995cba8c330b6463d628921","observation_id":"ca240d33-c54f-4067-89ec-dc6749ba7bd9","resolution":{"observed_at":"2026-05-10T00:14:46.454940Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2604.24558","last_updated":"2026-04-27T14:47:22Z","snapshot_observed_at":"2026-07-06T23:10:33.821313Z","submitted_at":"2026-04-27T14:47:22Z","title":"Hierarchical Behaviour Spaces","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-08T03:33:48.527621Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2604.24558"},"observation_digest":"sha256:9b3332d9b8a06632377e738f6014e95a10325dc71f959660729b3556b3b5eb53","observation_id":"95274a99-b203-4a2b-97fa-9ba2d1802eb3","resolution":{"observed_at":"2026-05-11T22:01:12.700310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.06869","last_updated":"2026-05-12T18:33:33Z","snapshot_observed_at":"2026-07-31T07:18:07.162544Z","submitted_at":"2026-05-07T19:12:03Z","title":"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-11T01:22:32.713175Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.06869"},"observation_digest":"sha256:e97c1489734ef5e19fb5803ea65dec0a57ea9ab387fd1a67f3ad071a8ad1af9d","observation_id":"dbd0bfa9-aaea-431a-a21c-32d6c9e874ee","resolution":{"observed_at":"2026-05-11T04:25:58.628439Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.06869","last_updated":"2026-05-12T18:33:33Z","snapshot_observed_at":"2026-07-31T07:18:07.162544Z","submitted_at":"2026-05-07T19:12:03Z","title":"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-14T20:51:26.063471Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.06869"},"observation_digest":"sha256:01e98bd9b8ac0a365336c3d7271154b6f1c82f6a39e14af8938fe443b946b5ad","observation_id":"de8bb243-1d72-4652-9007-582c9d8d1060","resolution":{"observed_at":"2026-05-14T20:52:57.892871Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.07021","last_updated":"2026-05-20T02:19:39Z","snapshot_observed_at":"2026-08-02T16:20:14.264105Z","submitted_at":"2026-05-07T23:05:50Z","title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-11T00:54:25.549158Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.07021"},"observation_digest":"sha256:8b13da43e67f96769ef7afdfceff0f3f10cf233880fec0cd457cb28646b62d67","observation_id":"dac7dc5c-5c4d-4969-961b-e11ebdffe1e2","resolution":{"observed_at":"2026-05-11T05:00:56.528187Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.07021","last_updated":"2026-05-20T02:19:39Z","snapshot_observed_at":"2026-08-02T16:20:14.264105Z","submitted_at":"2026-05-07T23:05:50Z","title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T08:29:09.122055Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.07021"},"observation_digest":"sha256:c6f063b9177147b55108df2f7cfe02eb8cfc8f6c9895bf06d994958238549c9e","observation_id":"8960fb7d-4e4f-4233-aa2d-c03a3bfc03aa","resolution":{"observed_at":"2026-05-21T08:29:52.732321Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.09965","last_updated":"2026-05-12T15:54:46Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:16:41Z","title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","version":1},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-05-12T03:25:24.844859Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.09965"},"observation_digest":"sha256:4d90582054a5914b79ed26ede53005988a27f4063a3ae5ff4ab845a61e00edb6","observation_id":"db4efc6b-1dc2-4fe6-8df1-481204036066","resolution":{"observed_at":"2026-05-12T03:26:19.238017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.09965","last_updated":"2026-05-12T15:54:46Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:16:41Z","title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","version":2},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-05-13T06:44:28.552513Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.09965"},"observation_digest":"sha256:a889590967fb6d44f26f09640f0b3d5669e7fedbfe4cd809337d40b78d96439c","observation_id":"08a1e57d-28ee-429e-9add-6411e0557f3f","resolution":{"observed_at":"2026-05-13T06:47:27.232473Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.11223","last_updated":"2026-05-11T20:33:48Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:33:48Z","title":"Do Vision-Language-Models show human-like logical problem-solving capability in point and click puzzle games?","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-13T01:58:39.476408Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.11223"},"observation_digest":"sha256:45bf975f1e1ae671806e2b77fc2aa0e5d7a7e4e7d2ce376394db59276079bb87","observation_id":"e32d29a5-3527-4721-b0f1-63d19cbcfee8","resolution":{"observed_at":"2026-05-13T02:02:05.521656Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.11223","last_updated":"2026-05-11T20:33:48Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:33:48Z","title":"Do Vision-Language-Models show human-like logical problem-solving capability in point and click puzzle games?","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-13T01:58:39.476408Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.11223"},"observation_digest":"sha256:56a5eb95eb17efd620d58f7eb875372a0c2d9a60d6e4f4f63e4087c88478bf41","observation_id":"3f1cd5a9-a3de-47d8-b583-e05ab03d8ada","resolution":{"observed_at":"2026-05-13T02:07:08.791635Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.13875","last_updated":"2026-05-08T06:56:35Z","snapshot_observed_at":"2026-08-02T22:27:33.584660Z","submitted_at":"2026-05-08T06:56:35Z","title":"Common-agency Games for Multi-Objective Test-Time Alignment","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-15T06:14:53.685486Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.13875"},"observation_digest":"sha256:f975ed3f3c69084f3d8cc5cec5da65e86b8533c6d8b544ad3b101d68a309a429","observation_id":"076b6c1a-aa9f-40c5-8d4b-97b2bb7c1fee","resolution":{"observed_at":"2026-05-15T06:15:06.433297Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.17933","last_updated":"2026-05-18T06:41:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-18T06:41:14Z","title":"AtlasVA: Self-Evolving Visual Skill Memory for Teacher-Free VLM Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T11:24:48.558423Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.17933"},"observation_digest":"sha256:2555cef445c94430a74e8bd9e663849a9998407f0b8d909f9201b27166743aff","observation_id":"3078f913-8c75-40fd-b21b-dd69bd2f9666","resolution":{"observed_at":"2026-05-20T11:28:14.500972Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2605.24539","last_updated":"2026-05-23T12:10:52Z","snapshot_observed_at":"2026-07-06T23:34:38.098785Z","submitted_at":"2026-05-23T12:10:52Z","title":"DemoEvolve: Overcoming Sparse Feedback in Agentic Harness Evolution with Demonstrations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-30T13:12:46.927103Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2605.24539"},"observation_digest":"sha256:8f436736a63f6d9452596c606b0ff479879d993239fdd0bedc45883df8b12622","observation_id":"9a439c1a-4186-4b59-bc87-810f66df3813","resolution":{"observed_at":"2026-06-30T13:14:40.644739Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2606.05896","last_updated":"2026-06-22T18:29:42Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T09:03:43Z","title":"Resonant Minds: Closed-Loop Social Avatars with Theory of Mind","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T02:04:39.753443Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2606.05896"},"observation_digest":"sha256:73f6f30dd3633a51807c1de81924bad803ecd194ab6567f0c42ddb6a89b0db8a","observation_id":"677af669-2782-48b4-85fa-82fe1bbc86c3","resolution":{"observed_at":"2026-07-02T12:36:56.229554Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2606.09826","last_updated":"2026-06-08T17:59:43Z","snapshot_observed_at":"2026-08-02T20:00:31.322130Z","submitted_at":"2026-06-08T17:59:43Z","title":"OmniGameArena: A Unified UE5 Benchmark for VLM Game Agents with Improvement Dynamics","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-06-27T16:50:36.194650Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2606.09826"},"observation_digest":"sha256:85f118905521b539114586a2e6728a6c7d21816b73f875d9167ccaaf9f600611","observation_id":"6419eb26-cc55-40a9-af78-ea8bd5f3c81a","resolution":{"observed_at":"2026-07-03T01:07:29.936088Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2606.12191","last_updated":"2026-06-10T15:15:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-10T15:15:01Z","title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","version":1},"reference_index":114,"source":"pdf_text","source_observed_at":"2026-06-27T09:46:30.702256Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2606.12191"},"observation_digest":"sha256:f60a1dcbecf846e841e3065a7383c9bb864554c7af524ffd0355074fcf05b261","observation_id":"36304c02-9475-4d6a-9927-07b9e08406fd","resolution":{"observed_at":"2026-07-03T10:58:02.898623Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2606.13608","last_updated":"2026-06-11T17:23:54Z","snapshot_observed_at":"2026-07-06T23:52:18.996096Z","submitted_at":"2026-06-11T17:23:54Z","title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T06:41:41.799596Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2606.13608"},"observation_digest":"sha256:ad9118b812d6dd972dc0a578802de712ed6dabe54f4d0d309bb6369c8f4f030f","observation_id":"7af6216c-f02c-45ab-bd11-77827173cf68","resolution":{"observed_at":"2026-07-03T15:08:33.114440Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2606.18950","last_updated":"2026-06-18T08:25:27Z","snapshot_observed_at":"2026-08-03T04:41:12.393312Z","submitted_at":"2026-06-17T11:32:51Z","title":"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T21:02:26.081166Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2606.18950"},"observation_digest":"sha256:36043c3d2ee5ba091daacd596ce270f6646e4b9bdb8b796288b13996d77e3b94","observation_id":"e19c1c44-97b3-4b3e-bd77-ee3fafaabd4b","resolution":{"observed_at":"2026-07-04T00:39:17.286611Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2606.19338","last_updated":"2026-06-17T17:59:34Z","snapshot_observed_at":"2026-07-06T23:54:37.596257Z","submitted_at":"2026-06-17T17:59:34Z","title":"Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-26T21:17:02.332687Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2606.19338"},"observation_digest":"sha256:e2c0f7e9eb6883f0cfd39f2b1bea13f63439b5753888d3c71080917d7a046f16","observation_id":"ddf4e3cd-fae7-4de7-9a7a-3ccc871109a4","resolution":{"observed_at":"2026-07-04T00:19:13.650685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":"2411.13543","doi":"10.48550/arxiv.2411.13543","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games","venue":null,"work_id":"7158f298-1661-4f1f-976e-a8b900cdc32f","year":2024},"citing_paper":{"arxiv_id":"2607.01224","last_updated":"2026-07-01T17:57:03Z","snapshot_observed_at":"2026-08-02T09:03:27.642845Z","submitted_at":"2026-07-01T17:57:03Z","title":"AutoMem: Automated Learning of Memory as a Cognitive Skill","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-02T12:09:32.053764Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2607.01224"},"observation_digest":"sha256:ce40b3298f4e098a0e052b5b23ca87142ebc500e241d6374e65d8d9106bc8e02","observation_id":"21f38e44-3766-45fb-99a2-d083342a44eb","resolution":{"observed_at":"2026-07-02T12:16:56.514755Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-01T10:58:16.786880Z","title":"Balrog: Bench- marking agentic llm and vlm reasoning on games,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.22732","last_updated":"2026-07-22T12:10:45Z","snapshot_observed_at":"2026-08-03T11:59:17.695078Z","submitted_at":"2026-07-22T12:10:45Z","title":"Spatial Reasoning in LLM Game Agents: Impact of Causal Context and Multi-Step Planning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T10:58:16.786880Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2607.22732"},"observation_digest":"sha256:b5baaf37f6845f39308a7470d9f7eeae4eae3c9e224bfd6052276cbd5520c9a3","observation_id":"a20c9f69-bbab-485a-8f42-99b3d89d43fd","resolution":{"observed_at":"2026-08-01T10:58:16.786880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-03T04:15:06.000334Z","title":"Carlo Romeo and Andrew D","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29577","last_updated":"2026-07-31T16:03:38Z","snapshot_observed_at":"2026-08-04T18:23:20.602514Z","submitted_at":"2026-07-31T16:03:38Z","title":"DungeonBench: A Benchmark for Rules-Rich Tactical Reasoning in Dungeons & Dragons Combat","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T04:15:06.000334Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2607.29577"},"observation_digest":"sha256:435a7405b8d3594ea6b25a614453050ccfe5f94f6b00036973cda4a1b36b2c7a","observation_id":"c3cfa10d-5cd9-48a2-8036-69dc7653fd14","resolution":{"observed_at":"2026-08-03T04:15:06.000334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2411.13543/citation-record","integrity":"/paper/2411.13543/integrity","json":"/paper/2411.13543/citation-record.json","paper":"/paper/2411.13543"},"outbound":[],"paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2411.13543."}