{"as_of":"2026-08-08T01:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7de63481d2c44219b28be16b1514c45fb079533e257b4734c31dd478a0a0ade8","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:50:09.623638Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T11:40:21.327031Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T11:40:25.831314Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"cited_work":{"arxiv_id":"2507.12391","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12391","snapshot_observed_at":"2026-08-05T11:40:25.831314Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","venue":"cs.RO","work_id":"e6e37e5d-76d1-4673-916d-5b43248e3707","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-07T13:45:24.073089Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.327031Z"},"links":{"cited_paper":"/paper/2507.12391","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:5f02e4a1b1873ee0daecc00ade3d72697c0c05bba2396c7659541f48793b0fd6","observation_id":"ff00f254-9fce-4018-87cd-dad413f06d7e","resolution":{"observed_at":"2026-08-05T11:40:25.837858Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.12391/citation-record","integrity":"/paper/2507.12391/integrity","json":"/paper/2507.12391/citation-record.json","paper":"/paper/2507.12391"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:12.589834Z","title":"A survey on large language model based autonomous agents","venue":null,"work_id":"7909e68d-290c-4bb2-9440-062c1fa095a8","year":2024},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:07.398980Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:a3153faa0eccd4b032e84c23bb9d7aef8b98064bfb5fdb2feabecd48fbfdceab","observation_id":"08cbf44a-4b00-4086-b09c-742698c10219","resolution":{"observed_at":"2026-08-06T16:50:12.672882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:12.454152Z","title":"Large language models for robotics: Opportunities, challenges, and perspectives","venue":null,"work_id":"bca01705-0f0c-47fb-b32e-0984b524a909","year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:07.474604Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:66243761bbf0abb9848f8f51387eb81e5a184b19654bd632dcbbc47e654df242","observation_id":"9986f59e-4f4c-475a-965f-7ee8a08b25c0","resolution":{"observed_at":"2026-08-06T16:50:12.514886Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:12.239608Z","title":"Human-Robot collaboration in surgery: Advances and challenges towards autonomous surgical assis- tants","venue":null,"work_id":"eb208020-b7a2-4b3b-a558-9852d8acb8bb","year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:07.565616Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:3781902c22a1df1b51a76e2cf68153a9a3a5cd9a1c6673fc2bb9113fbb3a59a2","observation_id":"c65b190b-166d-4fcc-a5f3-a3722221d776","resolution":{"observed_at":"2026-08-06T16:50:12.354229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:12.056072Z","title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","venue":null,"work_id":"7adf1e05-5525-4e03-a9f2-318dfd572337","year":2022},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:07.700942Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:0f5834f46bb248904b9a33d744d9e195dd16d8a479503b2e8c8703860d8e09b2","observation_id":"57623a7e-2b68-4061-abe0-f2ee19db3928","resolution":{"observed_at":"2026-08-06T16:50:12.147985Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:11.841297Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","venue":null,"work_id":"3d2c43b4-44e2-4d3b-be08-f2fe5fe85384","year":2023},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:07.858503Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:0d64a415e1fd60f6acee56334ca851ffde46997fa1be7175dc7553a53c7fa3e6","observation_id":"cbc1c722-8cdf-4b41-b90a-9a502e50e7a8","resolution":{"observed_at":"2026-08-06T16:50:11.952899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:11.548450Z","title":"Progprompt: Generating situated robot task plans using large language models","venue":null,"work_id":"95cc3882-94af-4752-aee1-c61db00f1e52","year":2023},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:07.985416Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:36edf74cec64a49884c1eb8f60a06b177e133f3b29f5d4b0731865bbc2c49234","observation_id":"3f2e3fba-f0c5-4623-88d5-2bbb2c9d94e3","resolution":{"observed_at":"2026-08-06T16:50:11.717772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:11.268373Z","title":"V oice con- trol interface for surgical robot assistants","venue":null,"work_id":"2bb69d77-de9e-49f1-affd-d7e65d2d3803","year":2024},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.092121Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:223c3a158aca588bfc769172683b191f27488d16609a1f2759b4b5c1927add7d","observation_id":"19a8e19c-e42c-4bdf-ab88-6750cce49d7a","resolution":{"observed_at":"2026-08-06T16:50:11.412453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:10.972221Z","title":"LLM-based ambiguity detection in natural language instructions for collaborative surgical robots","venue":null,"work_id":"9d1f1d30-e2e5-482b-b95f-e43b792ba0ac","year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.202134Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:629f4553088e5bd09e612db1ad1510b726ddf82fa59a5709dd883c438ad68a90","observation_id":"a94424e7-7caa-4cee-a8cf-f67ecc0d3653","resolution":{"observed_at":"2026-08-06T16:50:11.129833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:10.715812Z","title":"Toward autonomous robotic minimally invasive surgery: A hybrid framework combining task-motion planning and dynamic behavior trees","venue":null,"work_id":"94b06c3e-5c0e-4dd8-870f-bfaa0103c7c0","year":2023},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.392134Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:29db2aa69ccb09124ff15a17b701cebce7a9c85c839ce5ce61d8495cb5b763a2","observation_id":"1a42df65-51b0-4524-ba32-23ca0dd7a425","resolution":{"observed_at":"2026-08-06T16:50:10.835604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:10.552890Z","title":"Exploring embodied mul- timodal large models: Development, datasets, and future directions","venue":null,"work_id":"3d9895b6-1d71-48a3-b05c-9949af0efc60","year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.518679Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:dc24aa0e6e75b3ff069e1f9fdfa1c784dfcb1256dc1a3728d4cc58435b8cf855","observation_id":"4a39828d-d7e5-453d-88f5-03b56020e6eb","resolution":{"observed_at":"2026-08-06T16:50:10.620752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:08.597731Z","title":"Multimodal Fusion and Vision-Language Models: A Survey for Robot Vision","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.597731Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:0ef48f2b079947b336257c6c1af0b84bde497fae2b73e0fb59e5d242e5d9ea6c","observation_id":"47faf2df-fb07-430d-ad0a-1957377fb85c","resolution":{"observed_at":"2026-08-06T16:50:08.597731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-08-06T16:50:08.663316Z","title":"PaLM- E: An Embodied Multimodal Language Model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.663316Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:191e0ea9313d997f339db01d9a24987215d2a813521217e6dbdc761fce6446b9","observation_id":"16c1735e-0042-4033-abad-8427410ddd74","resolution":{"observed_at":"2026-08-06T16:50:08.663316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-06T16:50:08.750449Z","title":"Rt-2: Vision- language-action models transfer web knowledge to robotic control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.750449Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:2b8e6afa795b0f119989a7f8da95dad5ebc17ab9f3481a74f512ab16553670af","observation_id":"07c40737-2ab5-4210-8b4e-34d859072622","resolution":{"observed_at":"2026-08-06T16:50:08.750449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03094","last_updated":"2023-05-28T07:32:38Z","snapshot_observed_at":"2026-08-07T13:44:24.778030Z","submitted_at":"2022-10-06T17:50:11Z","title":"VIMA: General Robot Manipulation with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03094","snapshot_observed_at":"2026-08-06T16:50:08.815246Z","title":"VIMA: General Robot Manipulation with Multimodal Prompts","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.815246Z"},"links":{"cited_paper":"/paper/2210.03094","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:67aab607b60516be74d91aea425445299b60f3dfc6dbfc1f645b18af2148eb3c","observation_id":"3c1d1cae-4389-435f-a7d6-0e4ba6a3e83e","resolution":{"observed_at":"2026-08-06T16:50:08.815246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.02511","last_updated":"2025-04-09T17:34:52Z","snapshot_observed_at":"2026-07-06T18:40:31.586903Z","submitted_at":"2024-06-20T01:24:30Z","title":"LLM-A*: Large Language Model Enhanced Incremental Heuristic Search on Path Planning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.02511","snapshot_observed_at":"2026-08-06T16:50:08.926716Z","title":"LLM-A*: Large Language Model En- hanced Incremental Heuristic Search on Path Plan- ning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.926716Z"},"links":{"cited_paper":"/paper/2407.02511","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:cf6ac8496c728f5f1f5dea3acaa390d844836c14881859df72059ceaea0b9bce","observation_id":"f1d0268a-a77f-4a04-9f93-2964c0b7155d","resolution":{"observed_at":"2026-08-06T16:50:08.926716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.01236","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:10.255811Z","title":"LLM-Advisor: An LLM Benchmark for Cost-efficient Path Planning across Multiple Terrains","venue":null,"work_id":"7efbe3f8-eb63-40e5-a261-a0f57bbe3bcf","year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.019979Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:045befbe73a8a38710456532f837a19839dfc68465304892ca7df095adfaea65","observation_id":"3ccd6026-280f-407f-892c-740167a13455","resolution":{"observed_at":"2026-08-06T16:50:10.352632Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02655","last_updated":"2024-12-03T18:29:37Z","snapshot_observed_at":"2026-07-06T20:01:05.921085Z","submitted_at":"2024-12-03T18:29:37Z","title":"LLM-Enhanced Path Planning: Safe and Efficient Autonomous Navigation with Instructional Inputs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02655","snapshot_observed_at":"2026-08-06T16:50:09.102714Z","title":"LLM-Enhanced Path Planning: Safe and Efficient Autonomous Navi- gation with Instructional Inputs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.102714Z"},"links":{"cited_paper":"/paper/2412.02655","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:e01d3e3b7fe7ec729f625d17a74a391d8b6b9175d5ee394fbd1d56f54782c538","observation_id":"622f095f-1e64-4971-a277-1bd0feddc52a","resolution":{"observed_at":"2026-08-06T16:50:09.102714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20666","last_updated":"2025-03-11T23:45:58Z","snapshot_observed_at":"2026-07-06T19:40:30.194147Z","submitted_at":"2024-10-28T01:58:21Z","title":"Guide-LLM: An Embodied LLM Agent and Text-Based Topological Map for Robotic Guidance of People with Visual Impairments","version":2},"cited_work":{"arxiv_id":"2410.20666","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.20666","snapshot_observed_at":"2026-08-06T16:50:10.006242Z","title":"Guide-LLM: An Embodied LLM Agent and Text-Based Topological Map for Robotic Guidance of People with Visual Impairments","venue":"cs.RO","work_id":"47fd81b6-168f-4535-9070-ae36f84e43cd","year":2024},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.172796Z"},"links":{"cited_paper":"/paper/2410.20666","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:2bbf74a06e309f955ecdcde5005af54efe451fec4e84524e111831826ea4000e","observation_id":"235df9a0-fdba-4ab2-a74f-909b7a894a4e","resolution":{"observed_at":"2026-08-06T16:50:10.109881Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16690","last_updated":"2025-02-23T19:09:01Z","snapshot_observed_at":"2026-08-07T17:54:49.268141Z","submitted_at":"2025-02-23T19:09:01Z","title":"From Text to Space: Mapping Abstract Spatial Models in LLMs during a Grid-World Navigation Task","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.16690","snapshot_observed_at":"2026-08-06T16:50:09.264012Z","title":"From Text to Space: Mapping Ab- stract Spatial Models in LLMs during a Grid-World Navigation Task","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.264012Z"},"links":{"cited_paper":"/paper/2502.16690","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:4e5b9cf1f98bcbc7681fd8aa5ad98946a90d0374701fdd8e2b5b6a7c5947393c","observation_id":"acc50030-12b0-4839-a227-482856a712ca","resolution":{"observed_at":"2026-08-06T16:50:09.264012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15037","last_updated":"2025-06-03T08:29:23Z","snapshot_observed_at":"2026-08-07T16:00:36.254737Z","submitted_at":"2025-04-21T11:48:39Z","title":"Scaling and Beyond: Advancing Spatial Reasoning in MLLMs Requires New Recipes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15037","snapshot_observed_at":"2026-08-06T16:50:09.379020Z","title":"A Call for New Recipes to Enhance Spatial Reasoning in MLLMs","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.379020Z"},"links":{"cited_paper":"/paper/2504.15037","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:8b71bdca233d07a3f05d9ac953dae91dd444d7120dc98bc3756af2054b8c8cdb","observation_id":"eeccf834-057c-4c69-be2d-ae4783029d6c","resolution":{"observed_at":"2026-08-06T16:50:09.379020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.13055","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:50:09.821123Z","title":"Mitigating Cross-Modal Distraction and Ensuring Geometric Feasibility via Affordance- Guided, Self-Consistent MLLMs for Food Prepara- tion Task Planning","venue":null,"work_id":"c53c6fc1-a23d-4e74-a52d-4d7410bbb4b1","year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.470153Z"},"links":{"citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:7ae3642ecca2e044c00b4a21a7fd5bd075f7fef8afd9a2fee8aa1fb1aacf5c49","observation_id":"6367c078-f26e-4924-bb48-ed3c51739195","resolution":{"observed_at":"2026-08-06T16:50:09.900096Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03249","last_updated":"2025-02-24T00:58:13Z","snapshot_observed_at":"2026-08-05T14:15:00.395755Z","submitted_at":"2023-10-05T01:42:16Z","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03249","snapshot_observed_at":"2026-08-06T16:50:09.535109Z","title":"Can Large Lan- guage Models be Good Path Planners? A Bench- mark and Investigation on Spatial-temporal Reason- ing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.535109Z"},"links":{"cited_paper":"/paper/2310.03249","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:b113ec0657063aca043de281767e1b780b4511e7c77adf9babafdabf160e7242","observation_id":"7f6edbf4-224d-4902-ae90-0d0f32c2d5d7","resolution":{"observed_at":"2026-08-06T16:50:09.535109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09560","last_updated":"2025-06-05T07:22:50Z","snapshot_observed_at":"2026-07-06T20:36:12.548343Z","submitted_at":"2025-02-13T18:11:34Z","title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.09560","snapshot_observed_at":"2026-08-06T16:50:09.623638Z","title":"Embodied- Bench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embod- ied Agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:09.623638Z"},"links":{"cited_paper":"/paper/2502.09560","citing_paper":"/paper/2507.12391"},"observation_digest":"sha256:6e8c77ed23b5cf04a5203f99d61878d9169dfde5bc2267968de7275a8cb5ef81","observation_id":"20a633f0-d5a2-4c3f-93ba-942cbab5df6d","resolution":{"observed_at":"2026-08-06T16:50:09.623638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","latest_version":1,"primary_category":"cs.RO","snapshot_observed_at":"2026-08-07T13:45:39.610114Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":3,"verified_fuzzy":10},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 1 inbound Pith citation observation for arXiv:2507.12391."}