{"as_of":"2026-08-20T00:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8556f99df33eb6e5513452f08d89197f003a16fddd6ac28fae69a02c77fc1d79","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:12:00.512507Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":50,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2304.11477","last_updated":"2023-09-27T07:29:44Z","snapshot_observed_at":"2026-08-12T21:21:48.006892Z","submitted_at":"2023-04-22T20:34:03Z","title":"LLM+P: Empowering Large Language Models with Optimal Planning Proficiency","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T18:36:18.342052Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2304.11477"},"observation_digest":"sha256:f65781596378a9d16f8701617b9995715ea55121ac137ebf754a19b5df9993e7","observation_id":"46c28c21-9af8-4463-b3f1-f41756e9b448","resolution":{"observed_at":"2026-05-14T18:36:18.607331Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2305.02301","last_updated":"2023-07-05T16:59:31Z","snapshot_observed_at":"2026-08-17T13:50:40.586963Z","submitted_at":"2023-05-03T17:50:56Z","title":"Distilling Step-by-Step! Outperforming Larger Language Models with Less Training Data and Smaller Model Sizes","version":2},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-05-21T20:50:09.265838Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2305.02301"},"observation_digest":"sha256:ee6bfddbbcd76c28ae934f3455099bb96ab424328d5366e3b64cf7b23daf61cd","observation_id":"bbaa211d-562c-4d18-983e-97e9def07777","resolution":{"observed_at":"2026-05-21T20:50:09.482935Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2305.14992","last_updated":"2023-10-23T07:24:28Z","snapshot_observed_at":"2026-08-15T11:39:07.998070Z","submitted_at":"2023-05-24T10:28:28Z","title":"Reasoning with Language Model is Planning with World Model","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-17T01:49:28.796581Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2305.14992"},"observation_digest":"sha256:5b0e32b0a9e4f465edc62f1ff3ed5431238f06d2ecf9c7835d61e3d3ef0ce714","observation_id":"2c3f695a-f259-42a7-aab1-b7dac58cbd74","resolution":{"observed_at":"2026-05-17T01:49:28.870191Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2309.02427","last_updated":"2024-03-15T15:44:11Z","snapshot_observed_at":"2026-08-18T07:01:35.707347Z","submitted_at":"2023-09-05T17:56:20Z","title":"Cognitive Architectures for Language Agents","version":3},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-16T19:33:44.146134Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2309.02427"},"observation_digest":"sha256:2d33b9e4825b3f327552db0d500e5a59a27b4c90a947793a1b4fa670c109e166","observation_id":"a17288dc-56c7-4a4c-98f2-39360714a03e","resolution":{"observed_at":"2026-05-16T19:33:44.195442Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2402.09664","last_updated":"2026-04-07T05:36:18Z","snapshot_observed_at":"2026-07-06T17:30:25.359458Z","submitted_at":"2024-02-15T02:24:46Z","title":"CodeMind: Evaluating Large Language Models for Code Reasoning","version":6},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-24T03:53:55.964755Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2402.09664"},"observation_digest":"sha256:4f0686ea592e2251816d0492f251e5be2c5e6a9fe449f304b7cd5689aa79bb67","observation_id":"458aa1fb-88d4-41bd-ba59-2b94e36d04d1","resolution":{"observed_at":"2026-05-24T03:55:59.764519Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2410.05229","last_updated":"2025-08-27T16:24:39Z","snapshot_observed_at":"2026-08-16T08:54:56.543625Z","submitted_at":"2024-10-07T17:36:37Z","title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-15T00:42:11.891829Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2410.05229"},"observation_digest":"sha256:81eee572da158ba3e2ee1d8adad4b3b3acabbae01a057eebacab2bb83fcf0dc3","observation_id":"72e0e72d-42d8-4193-93b0-f6eec1c4488e","resolution":{"observed_at":"2026-05-15T00:42:11.987226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-12T17:47:27.958065Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.12274","last_updated":"2024-11-19T06:53:54Z","snapshot_observed_at":"2026-08-16T23:58:20.705339Z","submitted_at":"2024-11-19T06:53:54Z","title":"A Review on Generative AI Models for Synthetic Medical Text, Time Series, and Longitudinal Data","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-12T17:47:27.958065Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2411.12274"},"observation_digest":"sha256:8ceacca99d1d3dac086767699028c8c65d794f3c1643a01852749419ad87d089","observation_id":"3e779026-5705-4a77-815d-9cbe0a9710eb","resolution":{"observed_at":"2026-08-12T17:47:27.958065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-11T06:01:13.407994Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.17874","last_updated":"2025-02-09T16:39:50Z","snapshot_observed_at":"2026-08-16T23:58:19.700639Z","submitted_at":"2024-12-22T09:10:34Z","title":"Evaluating LLM Reasoning in the Operations Research Domain with ORQA","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T06:01:13.407994Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2412.17874"},"observation_digest":"sha256:0e6821f707b856925a54e04ca387205077ab00a7a8b5c8d5cb4344bea6bd785f","observation_id":"fdac3854-9208-4f7e-b8a9-0ce2b26796fe","resolution":{"observed_at":"2026-08-11T06:01:13.407994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-09T23:21:56.512878Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.18482","last_updated":"2025-01-30T16:56:08Z","snapshot_observed_at":"2026-08-14T02:00:22.714318Z","submitted_at":"2025-01-30T16:56:08Z","title":"A Tool for In-depth Analysis of Code Execution Reasoning of Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-09T23:21:56.512878Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2501.18482"},"observation_digest":"sha256:04b636733a74437b32e092176f6b5904563edae0f1bfd73ceb5af109ed0a7a16","observation_id":"6c337865-8896-49b5-8a4f-99a5a4cdd197","resolution":{"observed_at":"2026-08-09T23:21:56.512878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-09T22:34:42.282181Z","title":"O.; Sreedharan, S.; and Kambhampati, S","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.18784","last_updated":"2026-07-08T10:18:17Z","snapshot_observed_at":"2026-08-15T18:12:56.314094Z","submitted_at":"2025-01-30T22:21:12Z","title":"Successor-Generator Planning with LLM-generated Heuristics","version":5},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-09T22:34:42.282181Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2501.18784"},"observation_digest":"sha256:bb6fd020d6e0e30c75779b1b27424c6dcd2942d7c4e8f1390ef5078aef2dd9c5","observation_id":"fafff5a4-8a2a-4493-9f1c-9db9facc9e43","resolution":{"observed_at":"2026-08-09T22:34:42.282181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2503.18018","last_updated":"2026-04-08T07:40:49Z","snapshot_observed_at":"2026-08-16T23:56:31.072762Z","submitted_at":"2025-03-23T10:35:39Z","title":"Lost in Cultural Translation: Do LLMs Struggle with Math Across Cultural Contexts?","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T22:27:02.059459Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2503.18018"},"observation_digest":"sha256:516926e1f674f2986bca96ad69cf4c312a0d9f68ed06132219113ade3f10d1f1","observation_id":"974da3df-3e6e-49d1-a9d6-e115a0e0bc69","resolution":{"observed_at":"2026-05-22T22:27:12.365359Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-16T10:12:00.507239Z","title":"Large language models still can’t plan (A benchmark for llms on planning and reasoning about change),","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18838","last_updated":"2025-04-26T07:48:52Z","snapshot_observed_at":"2026-08-17T04:06:12.060099Z","submitted_at":"2025-04-26T07:48:52Z","title":"Toward Generalizable Evaluation in the LLM Era: A Survey Beyond Benchmarks","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-08-16T10:12:00.507239Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2504.18838"},"observation_digest":"sha256:ecbdadf70b21ee43c4c44abcdd156d0b5df62b931eee0103f8fc8522dd35177c","observation_id":"96af8fdd-1d7b-41c9-be48-323a75670dcb","resolution":{"observed_at":"2026-08-16T10:12:00.507239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-16T10:12:00.512507Z","title":"Available: https://doi.org/10.48550/ arXiv.2206.10498","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18838","last_updated":"2025-04-26T07:48:52Z","snapshot_observed_at":"2026-08-17T04:06:12.060099Z","submitted_at":"2025-04-26T07:48:52Z","title":"Toward Generalizable Evaluation in the LLM Era: A Survey Beyond Benchmarks","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-16T10:12:00.512507Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2504.18838"},"observation_digest":"sha256:77554fafefbe1e901f2d399238364ccb50f48ee2559c0bdf7e54a42e22e52e7d","observation_id":"70585823-cd63-4605-b721-20a1a7f3b1ec","resolution":{"observed_at":"2026-08-16T10:12:00.512507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2506.06941","last_updated":"2025-11-20T00:19:24Z","snapshot_observed_at":"2026-08-09T21:19:24.140229Z","submitted_at":"2025-06-07T22:42:29Z","title":"The Illusion of Thinking: Understanding the Strengths and Limitations of Reasoning Models via the Lens of Problem Complexity","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-15T16:10:31.440921Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2506.06941"},"observation_digest":"sha256:f3f4dcc5540bcac997903ddce986e914906ecf6083f40653ff8eb631e293b90e","observation_id":"8588f832-6e54-4e86-a15a-c5f113590284","resolution":{"observed_at":"2026-05-15T16:10:31.560405Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-07T04:18:26.474007Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change).arXiv preprint arXiv:2206.10498, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.10897","last_updated":"2025-06-12T17:02:27Z","snapshot_observed_at":"2026-08-18T08:36:03.022205Z","submitted_at":"2025-06-12T17:02:27Z","title":"GenPlanX. Generation of Plans and Execution","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T04:18:26.474007Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2506.10897"},"observation_digest":"sha256:beecb7bba2e8a6356783e59883424cf30e174187b41cc0a2da06ac30f6c92552","observation_id":"3f61024a-0ca6-4b6f-a3fe-7ec72afc538f","resolution":{"observed_at":"2026-08-07T04:18:26.474007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-06T18:48:17.855773Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07302","last_updated":"2025-07-09T22:01:32Z","snapshot_observed_at":"2026-08-17T03:49:45.399820Z","submitted_at":"2025-07-09T22:01:32Z","title":"Application of LLMs to Multi-Robot Path Planning and Task Allocation","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T18:48:17.855773Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2507.07302"},"observation_digest":"sha256:8d6e6418b7c3277815887c0f8d5d17ad5e6e6ad065d111a3d6cb8fa6cdba43eb","observation_id":"d2e22f05-25ae-4671-b40e-d5b1c81925dc","resolution":{"observed_at":"2026-08-06T18:48:17.855773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-06T11:08:10.452864Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.23135","last_updated":"2025-07-30T22:30:48Z","snapshot_observed_at":"2026-08-16T23:58:10.722478Z","submitted_at":"2025-07-30T22:30:48Z","title":"ISO-Bench: Benchmarking Multimodal Causal Reasoning in Visual-Language Models through Procedural Plans","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T11:08:10.452864Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2507.23135"},"observation_digest":"sha256:bdd2e1d3991929c89aa92ebd9c8cae0465624fb3551326a8d4d8251d1493b9f5","observation_id":"1f25c7e5-5586-4b32-85a8-31548ce6f10d","resolution":{"observed_at":"2026-08-06T11:08:10.452864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2510.15079","last_updated":"2026-04-07T05:37:01Z","snapshot_observed_at":"2026-08-11T01:05:09.612408Z","submitted_at":"2025-10-16T18:48:12Z","title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-18T05:59:00.400429Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2510.15079"},"observation_digest":"sha256:175e32983e5164525e07bd631113f80a79c1b7afd396f1a7348b8fbf9c538451","observation_id":"d73f5b39-ae97-4245-838c-a31c63948ff2","resolution":{"observed_at":"2026-05-18T06:00:57.198770Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2605.05216","last_updated":"2026-04-17T01:45:30Z","snapshot_observed_at":"2026-08-16T04:32:49.951666Z","submitted_at":"2026-04-17T01:45:30Z","title":"SAT: Sequential Agent Tuning for Coordinator Free Plug and Play Multi-LLM Training with Monotonic Improvement Guarantees","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T09:34:33.271537Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.05216"},"observation_digest":"sha256:64dd82ee072ffc4cef40ecdf3a0ee65fcb962198b469ba72fb0eadd92f2ae9a8","observation_id":"1dab2a04-e5c0-4f11-a94c-aa669c1db994","resolution":{"observed_at":"2026-05-10T09:38:42.239711Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2605.08904","last_updated":"2026-05-09T11:51:34Z","snapshot_observed_at":"2026-08-14T08:35:38.597227Z","submitted_at":"2026-05-09T11:51:34Z","title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-12T02:57:15.521594Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.08904"},"observation_digest":"sha256:20ad50a3794a628e0f05674b0f2e7703cdcc96d8af45ce79cd64ee4cb83b30e7","observation_id":"95108b2b-5ec8-4fbd-8306-a6b0d4747c0b","resolution":{"observed_at":"2026-05-12T03:01:18.575710Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2605.10516","last_updated":"2026-05-11T13:06:24Z","snapshot_observed_at":"2026-07-06T23:22:28.318343Z","submitted_at":"2026-05-11T13:06:24Z","title":"Consistency as a Testable Property: Statistical Methods to Evaluate AI Agent Reliability","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-12T04:41:15.286881Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.10516"},"observation_digest":"sha256:b72e6a67f86881df32b1112e6d43a780e0a2b041611fb2ea9cfda5f555403be3","observation_id":"55b9f9ec-cad0-4ceb-9d3d-29102fa8297e","resolution":{"observed_at":"2026-05-12T04:41:21.825043Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2605.15333","last_updated":"2026-05-14T18:56:06Z","snapshot_observed_at":"2026-08-16T07:34:44.842584Z","submitted_at":"2026-05-14T18:56:06Z","title":"Zero-Shot Goal Recognition with Large Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-19T16:09:12.822056Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.15333"},"observation_digest":"sha256:c6e76d7bd18649f8615660f79d48cef2d52d09003e015257c32dd459705f3f5a","observation_id":"d460edfd-ebd1-4424-a809-5fcbf51d7b1e","resolution":{"observed_at":"2026-05-19T16:12:39.235918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-04T05:05:38.669864Z","title":"arXiv:2206.10498 [cs]","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.24140","last_updated":"2026-08-02T16:07:48Z","snapshot_observed_at":"2026-08-16T23:43:13.158779Z","submitted_at":"2026-05-22T19:01:25Z","title":"HyperGuide: Hyperbolic Guidance for Efficient Multi-Step Reasoning in Large Language Models","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T05:05:38.669864Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.24140"},"observation_digest":"sha256:88b50eb0c7dfd4dee2e3c67500964f2f464507d54fe01cdb2d2698e4f9ff3693","observation_id":"7eb88bfb-b7f7-4f44-b3ce-5b81605174dc","resolution":{"observed_at":"2026-08-04T05:05:38.669864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2605.26333","last_updated":"2026-05-25T21:12:47Z","snapshot_observed_at":"2026-08-06T01:15:28.089602Z","submitted_at":"2026-05-25T21:12:47Z","title":"Managing Uncertainty in LLM-Generated Procedural Knowledge for Virtual Laboratory Planning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T21:21:45.412616Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.26333"},"observation_digest":"sha256:552160d2cf7424d6e9887001d7fefd051d9e8fcd463f3196da271b46a18190fa","observation_id":"1ff5f9a1-93ce-4990-9f19-baf398000339","resolution":{"observed_at":"2026-06-29T21:23:58.747294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2605.30052","last_updated":"2026-05-28T15:03:17Z","snapshot_observed_at":"2026-08-12T13:17:01.013411Z","submitted_at":"2026-05-28T15:03:17Z","title":"REPOT: Recoverable Program-of-Thought via Checkpoint Repair","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-06-29T06:22:46.440678Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2605.30052"},"observation_digest":"sha256:822e3be3c6c8f6638b7e1c3edc1030d9433bb784a5a2d81654f13547b4320e20","observation_id":"5e40099d-66c1-4566-bf85-85ca3420a0c6","resolution":{"observed_at":"2026-06-29T06:23:08.706412Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2606.22219","last_updated":"2026-06-20T20:41:43Z","snapshot_observed_at":"2026-08-12T14:12:46.701975Z","submitted_at":"2026-06-20T20:41:43Z","title":"Lost in Aggregation: A Multi-Scale Diagnostic Benchmark for LLM Spatial Navigation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-26T10:37:24.946718Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2606.22219"},"observation_digest":"sha256:f5ac0067cc662125ce15fb3c0c68c1c0bcfbda3f2c7e00189813dcf09bf0195e","observation_id":"ce0c0d53-c2c9-4b7d-b831-feb1191cd8a1","resolution":{"observed_at":"2026-06-26T10:39:18.328036Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":"2206.10498","doi":"10.48550/arxiv.2206.10498","metadata_source":"pith","pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models still can’t plan (a benchmark for llms on planning and reasoning about change)","venue":"cs.CL","work_id":"ac4a23f5-f300-4748-a75a-f0f29bcf6ee4","year":2022},"citing_paper":{"arxiv_id":"2607.07492","last_updated":"2026-07-08T14:53:55Z","snapshot_observed_at":"2026-08-16T07:40:00.537585Z","submitted_at":"2026-07-08T14:53:55Z","title":"Search, Fail, Recover: A Training Framework for Correction-Aware Reasoning","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-07-09T09:20:08.212837Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2607.07492"},"observation_digest":"sha256:fa40ac01d755797f8730a56cb882f20b47237c86ac27dfe097c6d34f634ba810","observation_id":"5d6a6729-a5ca-414c-aa49-15f52a989fe1","resolution":{"observed_at":"2026-07-09T09:26:08.694841Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:43.429028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10498","snapshot_observed_at":"2026-08-12T00:19:26.639704Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.08236","last_updated":"2026-08-08T17:05:08Z","snapshot_observed_at":"2026-08-18T15:16:31.844349Z","submitted_at":"2026-08-08T17:05:08Z","title":"LatticeMind: A Conflict-Aware Memory Primitive for Multi-Agent Systems","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-12T00:19:26.639704Z"},"links":{"cited_paper":"/paper/2206.10498","citing_paper":"/paper/2608.08236"},"observation_digest":"sha256:8a75c50f9962bd3b3e122a6163eacc244848569c11b38c1ddb3055bb700cf7c3","observation_id":"cbeebfab-0187-479c-a309-613c04f069b1","resolution":{"observed_at":"2026-08-12T00:19:26.639704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2206.10498/citation-record","integrity":"/paper/2206.10498/integrity","json":"/paper/2206.10498/citation-record.json","paper":"/paper/2206.10498"},"outbound":[],"paper":{"arxiv_id":"2206.10498","last_updated":"2023-11-26T01:15:41Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T23:57:18.971838Z","submitted_at":"2022-06-21T16:15:27Z","title":"PlanBench: An Extensible Benchmark for Evaluating Large Language Models on Planning and Reasoning about Change"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2206.10498."}