{"as_of":"2026-08-08T16:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8f640eb1c9bd99252491fcf84ebd45c37e99d3c21abb40fb2aace028a5dc8238","coverage":[{"denominator":21,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:40:02.915482Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T13:39:59.590925Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:59:43.323576Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-08-06T13:39:59.590925Z","title":"As LLMs increasingly serve as the foundation for autonomous agents (Duan et al., 2022), understanding their capacity for spatial rea- soningbecomescrucial","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.590925Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:dbdb59a80ad55d81c62048560202c5f4cff6dc84747c307574ce0911d0ca8e70","observation_id":"e40e9f1c-26bb-4f59-a470-b31aae587614","resolution":{"observed_at":"2026-08-06T13:39:59.590925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2605.09965","last_updated":"2026-05-12T15:54:46Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:16:41Z","title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-12T03:25:24.844859Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2605.09965"},"observation_digest":"sha256:f114e910d7956f210b8a32ed1be86ad02ef5b6d92437fead1e155695b2c013db","observation_id":"2bfd5544-d885-4fae-a784-63ccd80d8a5a","resolution":{"observed_at":"2026-05-12T03:26:18.991223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2605.09965","last_updated":"2026-05-12T15:54:46Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:16:41Z","title":"Towards Generalist Game Players: An Investigation of Foundation Models in the Game Multiverse","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T06:44:28.552513Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2605.09965"},"observation_digest":"sha256:bccca4941036e0348411cd88a3deaa1d8d858ad401023c5f29a7798bfcfa7d4a","observation_id":"8a1c1c7b-fa67-4df1-a8dd-2da3ab196fa4","resolution":{"observed_at":"2026-05-13T06:47:27.126726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2605.28277","last_updated":"2026-05-27T10:20:53Z","snapshot_observed_at":"2026-08-01T19:48:22.299809Z","submitted_at":"2026-05-27T10:20:53Z","title":"Do LLMs Build World Models From Text? A Multilingual Diagnostic of Spatial Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T12:02:14.499758Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2605.28277"},"observation_digest":"sha256:ea5ef641abc35d74edb3a13635842f58b732e3977aa3f91a1825260c88751b21","observation_id":"b3db04d6-9c50-4606-b4a3-b4e5bc6bb440","resolution":{"observed_at":"2026-06-29T12:03:23.712055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":"2507.20395","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-07-04T08:59:43.323576Z","title":"Mazeeval: A benchmark for testing sequential decision-making in language models","venue":null,"work_id":"529e0c95-b668-41c6-a159-f245a0c5365e","year":2025},"citing_paper":{"arxiv_id":"2606.22219","last_updated":"2026-06-20T20:41:43Z","snapshot_observed_at":"2026-07-06T23:57:09.273708Z","submitted_at":"2026-06-20T20:41:43Z","title":"Lost in Aggregation: A Multi-Scale Diagnostic Benchmark for LLM Spatial Navigation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T10:37:24.946718Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2606.22219"},"observation_digest":"sha256:25c644115d4f9875450ec3ede0ae548cc9a63899e74d5a3de0650721eb2cd511","observation_id":"c66416b3-7501-44d1-9685-c2c624d4d082","resolution":{"observed_at":"2026-07-04T08:59:43.325606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.20395/citation-record","integrity":"/paper/2507.20395/integrity","json":"/paper/2507.20395/citation-record.json","paper":"/paper/2507.20395"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.20395","snapshot_observed_at":"2026-08-06T13:39:59.590925Z","title":"As LLMs increasingly serve as the foundation for autonomous agents (Duan et al., 2022), understanding their capacity for spatial rea- soningbecomescrucial","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.590925Z"},"links":{"cited_paper":"/paper/2507.20395","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:dbdb59a80ad55d81c62048560202c5f4cff6dc84747c307574ce0911d0ca8e70","observation_id":"e40e9f1c-26bb-4f59-a470-b31aae587614","resolution":{"observed_at":"2026-08-06T13:39:59.590925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:09.464754Z","title":null,"venue":null,"work_id":"2e43031b-22f5-4cdb-93c3-97f2806bf265","year":2022},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.694094Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:9bb1a4f1c4c91b4fda5be9c09e59ada58933e4d0b96ac00f70df0a848487b9d5","observation_id":"99541be7-f270-4b91-aed6-9cbdbb8bbbcc","resolution":{"observed_at":"2026-08-06T13:40:09.594740Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.350782Z","title":"We focus on the configurationthatprovidesthemostchallengingyet fair assessment of spatial reasoning capabilities","venue":null,"work_id":"8469e64b-1557-4aa2-b0f4-4b8c119a6d3a","year":2023},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.204151Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:5edf8c29a59556ae57d76697a12ab043149b9119a3cf85f5bf0ba2b9d1bd6ae7","observation_id":"0d21d79d-f7d3-4b02-82d3-c505c491a83e","resolution":{"observed_at":"2026-08-06T13:40:08.444915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.086542Z","title":"The results re- veal significant variations in spatial reasoning capa- bilities and provide insights into how these abilities transfer across languages","venue":null,"work_id":"2265777b-904f-415a-b4fb-76e53fcd066d","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.343327Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:2785f7094a83eb02cc5e10822e6d69149d7ce4866093e649c82a4977f48a35c8","observation_id":"41bb9291-c22c-4056-b142-7a36ac46df0c","resolution":{"observed_at":"2026-08-06T13:40:08.254906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:07.784836Z","title":"illusion of thinking","venue":null,"work_id":"d7ee4d11-74b3-4339-b333-4bf6045ed2a5","year":2024},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.487347Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:009c581ade39fa2836e22d43b56491aacfc94b35599d0e73dc99d58e2d889053","observation_id":"434ccbdb-126d-429c-b277-edef3801563a","resolution":{"observed_at":"2026-08-06T13:40:07.924848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:06.985571Z","title":null,"venue":null,"work_id":"78aeacdc-cb93-4e5b-a5dc-3ac392f99e86","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.765301Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:8cbfb5efe04e23804149fd9bcc8e365bca7a9e40344ec433cb18b82a67ccc856","observation_id":"f50341c9-d0b7-4d17-ac02-04a6bfeb0dda","resolution":{"observed_at":"2026-08-06T13:40:07.194752Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:06.484759Z","title":null,"venue":null,"work_id":"87df8bf0-82ba-4f78-90b9-7a26cd9c612c","year":2025},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.958377Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:e7169908eaa261014e9024ae4f5a30ee91b8f386b1ddff75a7e2028554c4e62d","observation_id":"36f47d57-c02c-48d9-9ad0-6130d5b6d901","resolution":{"observed_at":"2026-08-06T13:40:06.673377Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03249","last_updated":"2025-02-24T00:58:13Z","snapshot_observed_at":"2026-08-05T14:15:00.395755Z","submitted_at":"2023-10-05T01:42:16Z","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03249","snapshot_observed_at":"2026-08-06T13:40:01.114830Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.114830Z"},"links":{"cited_paper":"/paper/2310.03249","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:69741f0bc20baf25530bb2adc7cdd8fe21a642c4272cf0dbd0fa8372dfee2d43","observation_id":"c22597bd-79f8-4566-8333-8d41d4303dd3","resolution":{"observed_at":"2026-08-06T13:40:01.114830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11164","last_updated":"2023-04-22T06:28:46Z","snapshot_observed_at":"2026-08-02T01:09:20.439387Z","submitted_at":"2023-04-22T06:28:46Z","title":"Dialectical language model evaluation: An initial appraisal of the commonsense spatial reasoning abilities of LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11164","snapshot_observed_at":"2026-08-06T13:40:01.425301Z","title":"Marc-Alexandre Côté, Akos Kádár, Xingdi Yuan, Ben Kybartas, Tavian Barnes, Emery Fine, James Moore, Matthew Hausknecht, Layla El Asri, Mahmoud Adada, et al","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.425301Z"},"links":{"cited_paper":"/paper/2304.11164","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:4b0a78f36fcfa787acb84f101839a9f1f02bed769e174ea21cd551d82eddb0f5","observation_id":"17eaa702-14d6-46c0-b172-226e24bd879a","resolution":{"observed_at":"2026-08-06T13:40:01.425301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.00530","last_updated":"2025-04-23T10:26:16Z","snapshot_observed_at":"2026-08-05T20:54:42.018363Z","submitted_at":"2023-11-01T14:08:56Z","title":"Advances in Embodied Navigation Using Large Language Models: A Survey","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.00530","snapshot_observed_at":"2026-08-06T13:40:01.594827Z","title":"In Proceedings of the AAAI Conference on Artificial Intelligence, volume 38, pages 18500–18507","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.594827Z"},"links":{"cited_paper":"/paper/2311.00530","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:b4ce7802fe3ee6e296eb1469ab62af46e8eea5def2a359db69eca06e23266cef","observation_id":"4d31b4ae-c2ec-4b68-988f-4c949a17cf66","resolution":{"observed_at":"2026-08-06T13:40:01.594827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01054","last_updated":"2023-12-02T07:41:46Z","snapshot_observed_at":"2026-08-05T20:45:16.610342Z","submitted_at":"2023-12-02T07:41:46Z","title":"Exploring and Improving the Spatial Reasoning Abilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.01054","snapshot_observed_at":"2026-08-06T13:40:01.783017Z","title":"InProceedings of the IEEE conferenceoncomputervisionandpatternrecog- nition, pages 8494–8502","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.783017Z"},"links":{"cited_paper":"/paper/2312.01054","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:f2c2c7d56e1756c8e57f44d1539be6f561fc5e25be1b7f3e7323f890c6532e40","observation_id":"f5ecded3-8aff-4c95-9681-4f77893aac7c","resolution":{"observed_at":"2026-08-06T13:40:01.783017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:06.084830Z","title":null,"venue":null,"work_id":"d32aa752-bca3-41c8-b473-f7ccd7915eea","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.184839Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:fc2c15cbef8983e3a86ec507eba8c39e89c6b8a5737fac442997f0195a75079c","observation_id":"1f793967-6f8e-417a-9031-75ebb38f585e","resolution":{"observed_at":"2026-08-06T13:40:06.244842Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:05.624951Z","title":null,"venue":null,"work_id":"15fb2b37-8e30-4fa4-a163-304089d58593","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.384763Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:07d9b1d64894de213f6b3939b24f77715e7ffbd537fe7fc299abb3fd2be1ea29","observation_id":"fcea007a-56e2-4d04-8b0a-8875684c0d63","resolution":{"observed_at":"2026-08-06T13:40:05.837369Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:05.135116Z","title":null,"venue":null,"work_id":"16840ffb-737a-45ea-96f4-c471baf68ff8","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.604877Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:8f3a526840136d5e12125a99fedbf3ac5acdb80e0881ffa11d21fa86d63540f2","observation_id":"d0996122-b4b5-4084-ac0b-27e838eba294","resolution":{"observed_at":"2026-08-06T13:40:05.374835Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:04.777170Z","title":null,"venue":null,"work_id":"691830e9-57e2-4cd8-8e17-103cdc1fe666","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.759852Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:88e5178efb899e76836acf1ddf8c6aad31a2eeb0695a52badfeba801d8bd064a","observation_id":"23523d3c-1ce1-4d82-a7d2-8e4b7cfd036d","resolution":{"observed_at":"2026-08-06T13:40:04.934759Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:04.385016Z","title":"Choose a direction: north, south, east, or west","venue":null,"work_id":"f1980128-40e6-4484-9a33-63be50720dc6","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:02.915482Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:2f505feb0e069f2e9ece7d55aa54f860c7ab5ff6240e043bd35c9f7129f11f09","observation_id":"7b9318bd-40fe-461a-8ce3-25d15b80fcc6","resolution":{"observed_at":"2026-08-06T13:40:04.577069Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1502.05698","last_updated":"2015-12-31T13:08:14Z","snapshot_observed_at":"2026-07-06T04:09:45.600447Z","submitted_at":"2015-02-19T20:46:10Z","title":"Towards AI-Complete Question Answering: A Set of Prerequisite Toy Tasks","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1502.05698","snapshot_observed_at":"2026-08-06T13:40:01.974813Z","title":"Jason Weston, Antoine Bordes, Sumit Chopra, Alexander M Rush, Bart Van Merriënboer, Ar- mand Joulin, and Tomas Mikolov","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2002,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.974813Z"},"links":{"cited_paper":"/paper/1502.05698","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:d61543a113355f308d580f61083140f4818f314e6af06a97311a4c9b142a3d7a","observation_id":"71c1483e-9315-4cdc-9085-a1149bcdf460","resolution":{"observed_at":"2026-08-06T13:40:01.974813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.984850Z","title":"TextWorld (Côté et al., 2018) offers text-based navigation but in richly described environments that provide substantial contextual cues","venue":null,"work_id":"748f1abd-451e-4822-b777-4e6499d8c24f","year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.837529Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:6e3703e7677c1e82ea15befc0a9912d94ed35ec6a912ba4f3a9b5bf3d27426e6","observation_id":"c45dc4ac-9566-44a9-a228-fe49abb6f589","resolution":{"observed_at":"2026-08-06T13:40:09.194750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:07.404863Z","title":null,"venue":null,"work_id":"565dbf0a-96a8-48fc-be23-bdf3434b6fea","year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:00.634891Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:1add507ffa0608b1d99844258f9c2d657d96b4d877874ee073f6da72fd686e30","observation_id":"a9adacda-781c-4bf2-975b-d2bcac3bfe2a","resolution":{"observed_at":"2026-08-06T13:40:07.565032Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T13:40:08.640129Z","title":null,"venue":null,"work_id":"fc5240db-c6df-490f-ab4a-53e70e225cd4","year":null},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:59.995856Z"},"links":{"citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:0ca95b3b79c9649b35c1c247e3e34be35caddec1e18f641e60fafa9d75777e3a","observation_id":"c2525298-d22f-4b43-ad67-4caf1691b83f","resolution":{"observed_at":"2026-08-06T13:40:08.774753Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.08272","last_updated":"2019-12-19T15:44:33Z","snapshot_observed_at":"2026-07-06T07:09:15.174187Z","submitted_at":"2018-10-18T20:48:08Z","title":"BabyAI: A Platform to Study the Sample Efficiency of Grounded Language Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.08272","snapshot_observed_at":"2026-08-06T13:40:01.294846Z","title":"In Proceedings of the IEEE/CVF Conference on ComputerVisionandPatternRecognition, pages 14455–14465","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T13:40:01.294846Z"},"links":{"cited_paper":"/paper/1810.08272","citing_paper":"/paper/2507.20395"},"observation_digest":"sha256:74a328414b5ee1639eeb6deddaad6618dc4cf81a73a728fef9407ecaa5d198a4","observation_id":"fa740c9c-f980-4591-a60b-95ef6745e02f","resolution":{"observed_at":"2026-08-06T13:40:01.294846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.20395","last_updated":"2025-07-27T19:33:45Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-06T13:39:58.565174Z","submitted_at":"2025-07-27T19:33:45Z","title":"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models"},"reference_resolution":{"displayed":21,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":5},"total_outbound_references":21},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 21 of 21 outbound references and 5 inbound Pith citation observations for arXiv:2507.20395."}