{"as_of":"2026-08-09T05:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4ec6297a02d0b10ba6c8543fdcfbe39f94d13c6ac10deeda2b871d85783e96d3","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T06:01:31.663090Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":10,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2312.13771","last_updated":"2026-07-05T07:51:04Z","snapshot_observed_at":"2026-08-04T19:11:48.797552Z","submitted_at":"2023-12-21T11:52:45Z","title":"AppAgent: Multimodal Agents as Smartphone Users","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-17T10:16:43.364787Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2312.13771"},"observation_digest":"sha256:9b773e6ab95eaa0f630c97ef59fbe1043c0a7ec1157136f9e561199ff83ce39d","observation_id":"f92ac27b-c38f-4428-957b-238844887057","resolution":{"observed_at":"2026-05-17T10:16:43.811973Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2401.05459","last_updated":"2024-05-08T06:16:23Z","snapshot_observed_at":"2026-08-02T13:57:57.119489Z","submitted_at":"2024-01-10T09:25:45Z","title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","version":2},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-05-17T00:57:26.303195Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2401.05459"},"observation_digest":"sha256:5ff964f4d1565e8e764e0742f4f427f4009e50a2902c667035d6e8d8ca85f0ac","observation_id":"4c803e01-53ff-42a6-a0c4-5c582bbebdee","resolution":{"observed_at":"2026-05-17T00:57:26.853627Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2401.10935","last_updated":"2024-02-23T04:36:51Z","snapshot_observed_at":"2026-08-08T03:28:12.161766Z","submitted_at":"2024-01-17T08:10:35Z","title":"SeeClick: Harnessing GUI Grounding for Advanced Visual GUI Agents","version":2},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-05-17T10:09:46.447508Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2401.10935"},"observation_digest":"sha256:cc230ed8d2191a6cce0c4bdd95f4c0c97afbab5fd4e49e23d20ad54aabbb03be","observation_id":"a2a67ec2-04ad-460f-8965-f01397d8fbff","resolution":{"observed_at":"2026-05-17T10:09:46.621740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-08-09T01:39:11.956106Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-13T01:19:32.406859Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2404.07972"},"observation_digest":"sha256:2721574a0e55fb3b1fbfb69e4b295a7e8ffe504b47a64fcdab0d0b261e9e8b66","observation_id":"1cbfd6e8-54ac-4c04-bb7a-bed3622f7ecf","resolution":{"observed_at":"2026-05-13T01:19:32.500326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2406.16173","last_updated":"2026-04-23T17:13:26Z","snapshot_observed_at":"2026-07-06T18:35:40.134423Z","submitted_at":"2024-06-23T17:53:10Z","title":"Crepe: A Mobile Screen Data Collector Using Graph Query","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-23T23:49:28.437024Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2406.16173"},"observation_digest":"sha256:237fbccee2dee1e091c5747359d36b1607d81adb5477b776d691c8e3b9822f49","observation_id":"70ba2b2e-0f73-47cb-b537-ed54a7729780","resolution":{"observed_at":"2026-05-23T23:53:38.367252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-07-06T19:57:55.925634Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"reference_index":160,"source":"pdf_text","source_observed_at":"2026-05-19T11:08:27.472508Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2411.18279"},"observation_digest":"sha256:b9f032360f0b5682db6bdfece9eece3224d75d747ace180d693fc9eb18c10734","observation_id":"f663f0d6-4877-4aad-b37f-e2144c41188f","resolution":{"observed_at":"2026-05-19T11:08:27.687253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2412.10345","last_updated":"2025-06-05T21:26:08Z","snapshot_observed_at":"2026-08-02T17:44:41.340688Z","submitted_at":"2024-12-13T18:40:51Z","title":"TraceVLA: Visual Trace Prompting Enhances Spatial-Temporal Awareness for Generalist Robotic Policies","version":3},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-15T18:27:22.760982Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2412.10345"},"observation_digest":"sha256:f3753ad87722a05ee824603ff6a25b6c496da3a6c32b9ae8e39f33c69efc885b","observation_id":"8999657e-da6c-48ec-9f81-5a68a0e8de2a","resolution":{"observed_at":"2026-05-15T18:27:23.011129Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2501.16150","last_updated":"2026-03-27T07:41:10Z","snapshot_observed_at":"2026-08-02T22:37:46.736145Z","submitted_at":"2025-01-27T15:44:02Z","title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","version":3},"reference_index":175,"source":"pdf_text","source_observed_at":"2026-05-23T04:59:36.994758Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2501.16150"},"observation_digest":"sha256:a7923286b977eba9cd4db0ea752294579d1e2521a77c1d58da85e44a696dd286","observation_id":"eb99b1c7-af39-479b-9897-0c5f6e2c9f47","resolution":{"observed_at":"2026-05-23T05:02:35.071271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-08T06:01:31.663090Z","title":"org/CorpusID:269042918","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08226","last_updated":"2025-02-14T06:23:57Z","snapshot_observed_at":"2026-08-09T00:37:19.405225Z","submitted_at":"2025-02-12T09:12:30Z","title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T06:01:31.663090Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2502.08226"},"observation_digest":"sha256:6d94fbc5505fb6fc58fb9d8812b61efd8bc3faed201a4b9c452d3a43cbddf89f","observation_id":"01fe3106-66a8-4a9a-86bd-b0d11aee631e","resolution":{"observed_at":"2026-08-08T06:01:31.663090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-07T20:57:48.531431Z","title":"Jianwei Yang, Hao Zhang, Feng Li, Xueyan Zou, Chunyuan Li, and Jianfeng Gao","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.15760","last_updated":"2025-02-13T18:55:14Z","snapshot_observed_at":"2026-08-07T22:34:53.289352Z","submitted_at":"2025-02-13T18:55:14Z","title":"Digi-Q: Learning Q-Value Functions for Training Device-Control Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T20:57:48.531431Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2502.15760"},"observation_digest":"sha256:f78dd4afdc19bfdcd7825353e9dd27b261d0b79175c67a47b195e68d3f6b9022","observation_id":"64465c25-1eda-44e2-9b03-6261ecdea371","resolution":{"observed_at":"2026-08-07T20:57:48.531431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2505.03364","last_updated":"2026-04-10T15:34:51Z","snapshot_observed_at":"2026-08-08T21:49:05.985084Z","submitted_at":"2025-05-06T09:37:51Z","title":"DroidRetriever: A Transparent and Steerable Automation System for Collaborative Mobile Information Seeking","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-22T17:03:27.724888Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2505.03364"},"observation_digest":"sha256:c6e59129549d678e4309426455a88f3f4db7add6cc744b676a9bfb5e619253f8","observation_id":"7fbeda4b-c123-434b-b167-a2f88d22c6c3","resolution":{"observed_at":"2026-05-22T17:05:00.424263Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2505.16120","last_updated":"2026-05-04T12:54:07Z","snapshot_observed_at":"2026-08-02T10:48:03.601811Z","submitted_at":"2025-05-22T01:52:15Z","title":"LLM-Powered AI Agent Systems and Their Applications in Industry","version":2},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-05-22T14:05:54.535411Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2505.16120"},"observation_digest":"sha256:814eaeab31a9b70c2c1b8459aa36fcbc2761d720b9ae94bbbec55c510189a1b6","observation_id":"fb8b5078-f6e8-4b17-a646-e9b6fe6caea6","resolution":{"observed_at":"2026-05-22T14:06:38.069372Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-07T11:11:48.250484Z","title":"Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03095","last_updated":"2025-06-03T17:27:04Z","snapshot_observed_at":"2026-08-07T11:06:35.897879Z","submitted_at":"2025-06-03T17:27:04Z","title":"DPO Learning with LLMs-Judge Signal for Computer Use Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:11:48.250484Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2506.03095"},"observation_digest":"sha256:bbc056070921a2d281b10e599f523b4ba02a4f534517843e5143b3e03e351d73","observation_id":"f3e53efa-5ea7-40c2-a989-a6f99ef929aa","resolution":{"observed_at":"2026-08-07T11:11:48.250484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-07T05:04:04.450958Z","title":"Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08972","last_updated":"2025-06-10T16:45:29Z","snapshot_observed_at":"2026-08-07T14:38:06.207367Z","submitted_at":"2025-06-10T16:45:29Z","title":"Atomic-to-Compositional Generalization for Mobile Agents with A New Benchmark and Scheduling System","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T05:04:04.450958Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2506.08972"},"observation_digest":"sha256:60da99059e72068b19b0728bae66c7af292674fe4c0177a4579736755ba22746","observation_id":"f8344364-2b2d-46e1-9bff-8d656eb6c471","resolution":{"observed_at":"2026-08-07T05:04:04.450958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T23:55:50.980394Z","title":"Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04700","last_updated":"2025-08-12T15:11:53Z","snapshot_observed_at":"2026-08-07T09:56:01.664660Z","submitted_at":"2025-08-06T17:58:46Z","title":"SEAgent: Self-Evolving Computer Use Agent with Autonomous Learning from Experience","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-05T23:55:50.980394Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2508.04700"},"observation_digest":"sha256:6186585c5c859b90e05cd590526ef47ffba88abf5d906ecd6a1d725f1938f002","observation_id":"ddf6b1b2-1d34-4457-8ee2-98f43820f179","resolution":{"observed_at":"2026-08-05T23:55:50.980394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T20:29:09.876871Z","title":"Yan et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.10955","last_updated":"2025-08-14T07:25:45Z","snapshot_observed_at":"2026-08-07T18:23:08.889281Z","submitted_at":"2025-08-14T07:25:45Z","title":"Empowering Multimodal LLMs with External Tools: A Comprehensive Survey","version":1},"reference_index":271,"source":"arxiv_source","source_observed_at":"2026-08-05T20:29:09.876871Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2508.10955"},"observation_digest":"sha256:9b94a26786b9e30fa899bfa5530d6d2c546716815289b7b570acb274f0d627ca","observation_id":"15d01908-2b6d-4437-a168-37e57d6c0471","resolution":{"observed_at":"2026-08-05T20:29:09.876871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T15:19:50.914458Z","title":"Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.20096","last_updated":"2025-08-27T17:59:50Z","snapshot_observed_at":"2026-08-05T15:19:32.265779Z","submitted_at":"2025-08-27T17:59:50Z","title":"CODA: Coordinating the Cerebrum and Cerebellum for a Dual-Brain Computer Use Agent with Decoupled Reinforcement Learning","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-05T15:19:50.914458Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2508.20096"},"observation_digest":"sha256:9fb9829a9d8a335d98ab64d23c8a423f208d8bb5b1e4887422e21685669c4ca0","observation_id":"5157dc22-57cf-42a6-9430-acbbbc010f77","resolution":{"observed_at":"2026-08-05T15:19:50.914458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-04T19:29:00.066242Z","title":"Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09265","last_updated":"2025-09-11T08:50:01Z","snapshot_observed_at":"2026-08-08T10:27:53.369430Z","submitted_at":"2025-09-11T08:50:01Z","title":"Harnessing Uncertainty: Entropy-Modulated Policy Gradients for Long-Horizon LLM Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T19:29:00.066242Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2509.09265"},"observation_digest":"sha256:1003fefe9d012ae0fe127540fbb75cb8bf6bf0f1d8b825a7f3ee35944450976f","observation_id":"79d06aa4-e3b2-4652-8bf4-333ada1bd8ef","resolution":{"observed_at":"2026-08-04T19:29:00.066242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-04T09:03:39.665469Z","title":"Gpt-4v in wonderland: Large multimodal models for zero-shot smartphone gui navigation.arXiv preprint arXiv:2311.07562,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.17790","last_updated":"2026-05-26T11:29:37Z","snapshot_observed_at":"2026-08-04T09:03:36.135394Z","submitted_at":"2025-10-20T17:48:26Z","title":"UltraCUA: A Foundation Model for Computer Use Agents with Hybrid Action","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T09:03:39.665469Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2510.17790"},"observation_digest":"sha256:9d928d6c16a81091d0446742703aa2d5410a299253289b6e465f5647ef2782cb","observation_id":"e74deca4-cee4-4534-923c-cd16a744070d","resolution":{"observed_at":"2026-08-04T09:03:39.665469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2512.10371","last_updated":"2026-05-08T09:40:02Z","snapshot_observed_at":"2026-07-06T22:38:40.044448Z","submitted_at":"2025-12-11T07:37:38Z","title":"AgentProg: Empowering Long-Horizon GUI Agents with Program-Guided Context Management","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T23:42:25.086602Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2512.10371"},"observation_digest":"sha256:abe052368901e03e6a3167cbf95e1371ca60a9d6f8085c5d85b399e8324f8c1a","observation_id":"f7ebcc2d-0478-4c6b-9dea-a556e6b2a206","resolution":{"observed_at":"2026-05-16T23:43:42.319081Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2604.23941","last_updated":"2026-04-27T01:29:02Z","snapshot_observed_at":"2026-08-02T10:14:35.344341Z","submitted_at":"2026-04-27T01:29:02Z","title":"GoClick: Lightweight Element Grounding Model for Autonomous GUI Interaction","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T04:46:22.676901Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2604.23941"},"observation_digest":"sha256:204c710e1b045052e980c9c5607fcf032673f19bbcd5a89649bf0f4a770eadb6","observation_id":"72f3147f-895d-465e-b6cd-5e477e90ea97","resolution":{"observed_at":"2026-05-11T21:36:17.216793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":1},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-05-07T06:30:09.945371Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:1ec720fa17bc5b23c742ef391315f91cbf9f0f755f7bb6472094fa38078f9111","observation_id":"01b321e9-f478-4689-852c-38ec73697524","resolution":{"observed_at":"2026-05-09T04:55:12.637531Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":2},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-05-08T03:12:19.414358Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:6cf4c774247926db25c22632737edf2f770f43e5b65fed03991b2fa359e7d893","observation_id":"73725326-d1d1-4fea-a304-2aa7b04c8828","resolution":{"observed_at":"2026-05-09T00:29:30.648701Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2604.27859","last_updated":"2026-05-15T06:25:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T13:43:25Z","title":"Rethinking Agentic Reinforcement Learning In Large Language Models","version":3},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-05-19T16:58:41.558250Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2604.27859"},"observation_digest":"sha256:ca0ce736a5641ed83b5decd4b88c231c31cc205098517678c71624db204bd90b","observation_id":"4379236c-47aa-4a7b-9c5d-fa57de81dc05","resolution":{"observed_at":"2026-05-19T17:02:40.468847Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2605.15542","last_updated":"2026-05-15T02:27:41Z","snapshot_observed_at":"2026-08-01T22:59:27.797060Z","submitted_at":"2026-05-15T02:27:41Z","title":"DRS-GUI: Dynamic Region Search for Training-Free GUI Grounding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-19T14:24:48.938948Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2605.15542"},"observation_digest":"sha256:2e46a059d87ad4ef62c6b899bc535a84ac9998837b8cd7c3d6732d2005e80872","observation_id":"efaddc4e-8616-499a-b53e-1829ee1c5085","resolution":{"observed_at":"2026-05-19T14:27:24.236214Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":"2311.07562","doi":"10.48550/arxiv.2311.07562","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gpt-4v in wonderland: Large multi- modal models for zero-shot smartphone gui navigation","venue":"arXiv (Cornell University)","work_id":"c31ca36a-fe7c-48e9-95e1-7f9bdf54ce89","year":2023},"citing_paper":{"arxiv_id":"2606.29705","last_updated":"2026-06-29T02:16:21Z","snapshot_observed_at":"2026-08-01T21:20:53.464369Z","submitted_at":"2026-06-29T02:16:21Z","title":"GUICrafter: Weakly-Supervised GUI Agent Leveraging Massive Unannotated Screenshots","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-30T06:39:12.591090Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2606.29705"},"observation_digest":"sha256:e2e54665e03ba48dfe84c41fdd34063b94270ab423bc25b34610487ffdc3c6a4","observation_id":"b0e70cfe-96a1-49e3-a311-fc5e95edbc2f","resolution":{"observed_at":"2026-06-30T06:44:19.201960Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-07-11T10:41:33.954036Z","title":"GPT-4V in wonderland: Large multimodal models for zero- shot smartphone GUI navigation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.04974","last_updated":"2026-07-06T12:04:49Z","snapshot_observed_at":"2026-08-03T03:48:32.816586Z","submitted_at":"2026-07-06T12:04:49Z","title":"A Comprehensive Study of Implementation Bugs in Multi-modal Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-11T10:41:33.954036Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2607.04974"},"observation_digest":"sha256:e3f3ebb8e9618a204f77b0518bce9c7619a716c078e564596f1f08cb3a8261be","observation_id":"371fd33c-1012-40a5-9904-19bbd25c4142","resolution":{"observed_at":"2026-07-11T10:41:33.954036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07562","snapshot_observed_at":"2026-08-01T16:20:11.333438Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.18046","last_updated":"2026-07-20T15:16:59Z","snapshot_observed_at":"2026-08-08T11:26:06.368477Z","submitted_at":"2026-07-20T15:16:59Z","title":"SEE: Structure-aware Exploring \\& Exploiting for Long-horizon GUI Agent Trajectory Synthesis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-01T16:20:11.333438Z"},"links":{"cited_paper":"/paper/2311.07562","citing_paper":"/paper/2607.18046"},"observation_digest":"sha256:7cea4986672af145a9a9e3d3ab85fb7fd010b1a924235cc4e2718ea91ed1f5a0","observation_id":"7e946baf-6ffa-4afc-bbf2-b68269c5e12c","resolution":{"observed_at":"2026-08-01T16:20:11.333438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.07562/citation-record","integrity":"/paper/2311.07562/integrity","json":"/paper/2311.07562/citation-record.json","paper":"/paper/2311.07562"},"outbound":[],"paper":{"arxiv_id":"2311.07562","last_updated":"2023-11-13T18:53:37Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:53:37Z","title":"GPT-4V in Wonderland: Large Multimodal Models for Zero-Shot Smartphone GUI Navigation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2311.07562."}