{"as_of":"2026-08-04T22:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c58cb1705ddca31d3d012809e83f04cddddd0b87038d326b291b0d004bcd9e6d","coverage":[{"denominator":12,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T21:09:01.397691Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":45,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T00:30:23.903148Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2410.23218","last_updated":"2024-10-30T17:10:19Z","snapshot_observed_at":"2026-07-30T04:21:00.152864Z","submitted_at":"2024-10-30T17:10:19Z","title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","version":1},"reference_index":143,"source":"arxiv_source","source_observed_at":"2026-05-13T09:29:27.173784Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2410.23218"},"observation_digest":"sha256:f631f19a5e113300c15047af857cb97cee1308b819bbea8ba70992fd8ec2d918","observation_id":"3af2819e-005b-4414-8ffb-84527da393ff","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-07-06T19:57:55.925634Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"reference_index":221,"source":"pdf_text","source_observed_at":"2026-05-19T11:08:27.472508Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2411.18279"},"observation_digest":"sha256:57fdc7bd36a3612ada2142ab40bbfb51fddc7c03e82effd32a7224b965e9294d","observation_id":"9cd062cd-2ed6-416c-b485-0849ec3a3d64","resolution":{"observed_at":"2026-05-19T11:08:27.789657Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2412.04454","last_updated":"2025-05-05T16:17:20Z","snapshot_observed_at":"2026-07-06T20:02:21.509050Z","submitted_at":"2024-12-05T18:58:26Z","title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-18T04:09:41.494136Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2412.04454"},"observation_digest":"sha256:05eec6ede7a3075a3d3af52c27ee575ab9875d35b7ae8f955012004fa945c180","observation_id":"b66e63d6-84e1-4fe0-8271-8d31b55fa58c","resolution":{"observed_at":"2026-05-18T04:09:41.561281Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2503.21620","last_updated":"2025-05-24T08:46:08Z","snapshot_observed_at":"2026-08-01T20:06:59.931737Z","submitted_at":"2025-03-27T15:39:30Z","title":"UI-R1: Enhancing Efficient Action Prediction of GUI Agents by Reinforcement Learning","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T11:02:41.335059Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2503.21620"},"observation_digest":"sha256:3ef0a640171848449582b11fc41ef6616ea39e6fb1d1c603554bc11fcf58f879","observation_id":"59f969bf-94ad-4151-b5eb-f5be48228c2b","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2504.10458","last_updated":"2025-10-01T04:55:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:45:54Z","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T02:10:57.976448Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2504.10458"},"observation_digest":"sha256:c28eb47b03399421f22248772895ef8575adab914a3db129904a8e5c1e38ea14","observation_id":"3bf6ad5f-2855-45b2-9878-8b28ae6aa2e5","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2504.14239","last_updated":"2025-04-19T09:25:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-19T09:25:55Z","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T13:54:44.011048Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2504.14239"},"observation_digest":"sha256:7b8fb83c64c4ba8fea0af7d4e0f6b01ee658e207e8406d7986b650fe8faa1af8","observation_id":"f2bfa13a-eb76-46bb-8eac-1062692c6e98","resolution":{"observed_at":"2026-05-18T13:54:44.259737Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2505.10887","last_updated":"2026-05-01T07:44:01Z","snapshot_observed_at":"2026-08-02T13:02:06.325821Z","submitted_at":"2025-05-16T05:43:27Z","title":"InfantAgent-Next: A Multimodal Generalist Agent for Automated Computer Interaction","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T15:18:15.475294Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2505.10887"},"observation_digest":"sha256:da3cdf156700ce72b9fa8743e5c3b4f7226e126d05c68b1c0f7c13d1020131fa","observation_id":"7a4ef3b9-8bac-40b9-a4d4-023d511b9779","resolution":{"observed_at":"2026-05-22T15:21:45.004039Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2507.04227","last_updated":"2026-04-14T14:47:25Z","snapshot_observed_at":"2026-08-01T14:12:08.736574Z","submitted_at":"2025-07-06T03:31:36Z","title":"Mobile GUI Agents under Real-world Threats: Are We There Yet?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-19T06:57:25.257775Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2507.04227"},"observation_digest":"sha256:4fce70806660e00df0cde710ebbca6bdc3ef1210bf013757e75b694e95aa4258","observation_id":"1390c1b9-1c5f-4943-b9d3-403d1d1f82c7","resolution":{"observed_at":"2026-05-19T07:02:07.940523Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2507.05791","last_updated":"2025-10-03T23:50:19Z","snapshot_observed_at":"2026-07-30T10:31:35.758747Z","submitted_at":"2025-07-08T08:52:18Z","title":"GTA1: GUI Test-time Scaling Agent","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T13:54:59.938216Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2507.05791"},"observation_digest":"sha256:d7b4e148594b3113e4a1356d87e743532f083474145bf100053be526b0e74948","observation_id":"bd5d6f2a-9076-4772-9855-9db2050c5a4b","resolution":{"observed_at":"2026-05-17T13:55:00.001322Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2509.06477","last_updated":"2026-04-15T11:26:55Z","snapshot_observed_at":"2026-07-06T22:25:43.838468Z","submitted_at":"2025-09-08T09:43:48Z","title":"MAS-Bench: A Unified Benchmark for Shortcut-Augmented Hybrid Mobile GUI Agents","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-18T18:42:29.744124Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2509.06477"},"observation_digest":"sha256:d692b4589d3ac6f51f5ffebe6d2e62bef6419c8f0b3f2a3279f949106be4fc6f","observation_id":"785429e3-aa08-4aac-bb38-364be644f97b","resolution":{"observed_at":"2026-05-18T18:42:48.184174Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-08-04T00:30:23.903148Z","title":"Navigating the digital world as humans do: Universal visual grounding for gui agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.00810","last_updated":"2026-06-30T10:43:39Z","snapshot_observed_at":"2026-08-04T00:30:22.339802Z","submitted_at":"2025-11-02T05:34:21Z","title":"GUI-AIMA: Aligning Intrinsic Multimodal Attention with a Context Anchor for GUI Grounding","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T00:30:23.903148Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2511.00810"},"observation_digest":"sha256:8c01327f897c95f8403fa869944cd12a7c16239ac471dae95f68e81f02957906","observation_id":"13f1da45-8b1e-4948-8401-922d1044be0b","resolution":{"observed_at":"2026-08-04T00:30:23.903148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-08-03T23:06:04.872520Z","title":"Navigating the digital world as humans do: Universal visual grounding for gui agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.07332","last_updated":"2026-06-09T21:30:32Z","snapshot_observed_at":"2026-08-03T23:05:58.992115Z","submitted_at":"2025-11-10T17:35:21Z","title":"Grounding Computer Use Agents on Human Demonstrations","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-03T23:06:04.872520Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2511.07332"},"observation_digest":"sha256:0d5de7ffc9c5d069d59473f6ca9ba616dc1e04feeb0317c1adfb41783eb6c6a3","observation_id":"a68ee014-a1ed-4347-9653-66f0c360157a","resolution":{"observed_at":"2026-08-03T23:06:04.872520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2512.19396","last_updated":"2026-04-10T06:20:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-22T13:42:18Z","title":"EchoTrail-GUI: Building Actionable Memory for GUI Agents via Critic-Guided Self-Exploration","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T20:44:14.354535Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2512.19396"},"observation_digest":"sha256:e4adc7d8bc9b961efe3ab1a6205bf15000f424f8c035f801282a05f5c0f401cb","observation_id":"8caafa32-f401-454a-bd23-cf3e6a4211ff","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2602.22942","last_updated":"2026-04-11T07:11:11Z","snapshot_observed_at":"2026-08-03T04:02:26.271081Z","submitted_at":"2026-02-26T12:34:57Z","title":"ClawMobile: Rethinking Smartphone-Native Agentic Systems","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T19:20:36.580925Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2602.22942"},"observation_digest":"sha256:b6acba2bd7fe21e70380ea471c27ccd4a569a893934ecbb298cb06904857f8b4","observation_id":"e00b898c-efcb-4f16-a36a-a6f79aa67de7","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2603.26041","last_updated":"2026-04-24T02:25:38Z","snapshot_observed_at":"2026-07-06T22:50:41.736941Z","submitted_at":"2026-03-27T03:21:19Z","title":"Rethinking Token Pruning for Historical Screenshots in GUI Visual Agents: Semantic, Spatial, and Temporal Perspectives","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-14T23:59:49.017251Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2603.26041"},"observation_digest":"sha256:851ddf1ba569f0366fa084278a42df2bc767c79dc59bb6c9221923381d1f0b0a","observation_id":"be1d68c9-d85c-4312-b1c1-c16a5c7446a2","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.13019","last_updated":"2026-05-28T06:47:16Z","snapshot_observed_at":"2026-07-12T21:00:51.745183Z","submitted_at":"2026-04-14T17:55:46Z","title":"PrecisionCUA: Iterative Visual Refinement for Pixel-Precise Cursor Grounding in Code Editors","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-10T15:57:32.504665Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.13019"},"observation_digest":"sha256:3945c30a64cbbd63ff7decfc02b510ffa03c6e4e32a35b4d71c3424e33e6607b","observation_id":"c183cc45-6fd7-4706-986a-47764dba4521","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-12T21:01:02.036287Z","title":"Navigating the digital world as humans do: Universal visual grounding for gui agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.13019","last_updated":"2026-05-28T06:47:16Z","snapshot_observed_at":"2026-07-12T21:00:51.745183Z","submitted_at":"2026-04-14T17:55:46Z","title":"PrecisionCUA: Iterative Visual Refinement for Pixel-Precise Cursor Grounding in Code Editors","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-07-12T21:01:02.036287Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.13019"},"observation_digest":"sha256:d962bd2a0f0ddba781593812a195e4893b037bc625e2b6256f0b36874f9ac66b","observation_id":"a82fd94a-15a3-4cca-86a7-fbb4a7378996","resolution":{"observed_at":"2026-07-12T21:01:02.036287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.13488","last_updated":"2026-04-15T05:23:04Z","snapshot_observed_at":"2026-08-02T15:35:18.284449Z","submitted_at":"2026-04-15T05:23:04Z","title":"Towards Scalable Lightweight GUI Agents via Multi-role Orchestration","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T13:45:17.686098Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.13488"},"observation_digest":"sha256:e57fd92bc2eaacf84d9d85529c5524d3299190e7fe00fda2fda26f100e658e5d","observation_id":"93db7a82-e381-4827-a1ff-fd2afdf3bebd","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.14113","last_updated":"2026-04-15T17:32:28Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:32:28Z","title":"UI-Zoomer: Uncertainty-Driven Adaptive Zoom-In for GUI Grounding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T14:06:55.472857Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.14113"},"observation_digest":"sha256:a3eeb26f75909d6f2bb0dd4ed0c8fd5f77cf36f359d8e29bf21b62282500e486","observation_id":"f53efa28-59b6-4fb0-8de7-a22d3fdb28cd","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.21268","last_updated":"2026-04-23T04:23:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-23T04:23:31Z","title":"Measure Twice, Click Once: Co-evolving Proposer and Visual Critic via Reinforcement Learning for GUI Grounding","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-09T23:05:05.251150Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.21268"},"observation_digest":"sha256:e0a9fdafb6c321a7d0223b8d3ab0dc90e7002509ce7b9b4ad543089f337c9182","observation_id":"22b0e0f3-6160-4d1d-b871-7bdb871b4f4b","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.25380","last_updated":"2026-05-08T04:29:20Z","snapshot_observed_at":"2026-07-06T23:11:16.247497Z","submitted_at":"2026-04-28T08:43:58Z","title":"Benchmarking and Improving GUI Agents in High-Dynamic Environments","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T16:55:29.332273Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.25380"},"observation_digest":"sha256:2ce15eec6917323d662aed84f8049a0183955dddd20357c93554fcbd2e1c0cb2","observation_id":"cc3a99d4-87a9-470e-a47f-c34aa2bf7019","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.25380","last_updated":"2026-05-08T04:29:20Z","snapshot_observed_at":"2026-07-06T23:11:16.247497Z","submitted_at":"2026-04-28T08:43:58Z","title":"Benchmarking and Improving GUI Agents in High-Dynamic Environments","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T00:54:49.351703Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.25380"},"observation_digest":"sha256:5b94a33714b34dda006899f064a4954733e3c955f7ff44db88e55a4eef1d53a7","observation_id":"429c4d92-3f10-4b8a-aaa3-c2460d09cfe4","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2604.27955","last_updated":"2026-04-30T14:51:49Z","snapshot_observed_at":"2026-07-06T23:13:20.516759Z","submitted_at":"2026-04-30T14:51:49Z","title":"GUI Agents with Reinforcement Learning: Toward Digital Inhabitants","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-07T05:48:00.486572Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2604.27955"},"observation_digest":"sha256:fbab51a187ada06b7cac7ed61a7a612e7bb46b1ee908112551d87e5904438e1d","observation_id":"abe7fd51-a051-467e-9bf7-f1a67f003b22","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.00642","last_updated":"2026-05-10T03:40:15Z","snapshot_observed_at":"2026-08-02T17:34:14.464282Z","submitted_at":"2026-05-01T13:23:26Z","title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-09T19:36:42.309114Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.00642"},"observation_digest":"sha256:d97c8619b9c357ea1b38713db04c143764fa7c3ed6a832eaa23f58b9fd1f6619","observation_id":"3ad663ce-706b-4bfd-a19d-60e19b23f7d0","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.00642","last_updated":"2026-05-10T03:40:15Z","snapshot_observed_at":"2026-08-02T17:34:14.464282Z","submitted_at":"2026-05-01T13:23:26Z","title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T02:01:51.292863Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.00642"},"observation_digest":"sha256:e069945723bf6df5077a98602dd0372f8bd1d0cdd326b235b30f1f775e67f15f","observation_id":"2e8df0c6-5ac9-47da-a749-e8ea8b98bd43","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.02630","last_updated":"2026-05-04T14:18:46Z","snapshot_observed_at":"2026-08-02T10:02:24.148777Z","submitted_at":"2026-05-04T14:18:46Z","title":"AutoFocus: Uncertainty-Aware Active Visual Search for GUI Grounding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T18:37:48.993080Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.02630"},"observation_digest":"sha256:13edab621a1f54cc63bb4d10b3175d046a31284e892a7cd70a8be777b64831aa","observation_id":"eafcaccd-6d04-45c8-97e0-d48db3c9e4bc","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.02730","last_updated":"2026-05-04T15:31:11Z","snapshot_observed_at":"2026-08-02T14:48:33.210016Z","submitted_at":"2026-05-04T15:31:11Z","title":"Perceptual Flow Network for Visually Grounded Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-08T18:40:55.753827Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.02730"},"observation_digest":"sha256:b83f857bc104781b2539fef1d1c9169b4d0c02c3ef3f1da74c10f3c9e77b897f","observation_id":"d66df7ea-1683-4d03-acc4-3baf5c1e37cb","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.08560","last_updated":"2026-05-08T23:41:13Z","snapshot_observed_at":"2026-08-02T10:00:31.466870Z","submitted_at":"2026-05-08T23:41:13Z","title":"ZAYA1-VL-8B Technical Report","version":1},"reference_index":156,"source":"pdf_text","source_observed_at":"2026-05-12T01:15:16.607346Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.08560"},"observation_digest":"sha256:b430c2d27a88c4b8803f8e9f2c3da4f9b8df8788d05448e29325e397599ce97a","observation_id":"19d9e1cc-7b24-40e2-bac5-05265d3a833d","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.12501","last_updated":"2026-05-12T17:59:58Z","snapshot_observed_at":"2026-08-02T22:39:49.566685Z","submitted_at":"2026-05-12T17:59:58Z","title":"Covering Human Action Space for Computer Use: Data Synthesis and Benchmark","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T05:18:23.310925Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.12501"},"observation_digest":"sha256:209968f11909f5fbd73f27e304167dbe2923e8b5cf6102bed37314f3f10a50d0","observation_id":"b8f18861-2bf5-4759-8e27-165840a68e8b","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T19:05:36.511150Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:baf6f21ec3c9cdcd377d2a26d85c54064e62b7eacc26fd187fdd06a22d808113","observation_id":"1d02c06b-62a2-45e8-94d6-a63b0285f518","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-15T05:59:44.669877Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:43c9e1d408e1de842d749e78296f0db8c832efed11e6834f677108096a5ea0f0","observation_id":"640bd270-257b-41c7-990a-dfcda80e26ab","resolution":{"observed_at":"2026-05-16T21:09:01.446022Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.13527","last_updated":"2026-06-01T11:38:10Z","snapshot_observed_at":"2026-07-06T23:25:07.022065Z","submitted_at":"2026-05-13T13:40:31Z","title":"MMSkills: Towards Multimodal Skills for General Visual Agents","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T21:31:20.079403Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.13527"},"observation_digest":"sha256:1bdfa40a1bda8c77311d243c10ecd24083000934c8aea61af83ca65a84d31c82","observation_id":"a5d0bb76-3bb8-4243-98c6-88069d3677a3","resolution":{"observed_at":"2026-06-30T21:35:04.554540Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.15542","last_updated":"2026-05-15T02:27:41Z","snapshot_observed_at":"2026-08-01T22:59:27.797060Z","submitted_at":"2026-05-15T02:27:41Z","title":"DRS-GUI: Dynamic Region Search for Training-Free GUI Grounding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-19T14:24:48.938948Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.15542"},"observation_digest":"sha256:8ed530f95d974e759a54414ddb3cadb038f3eac0bf85313b3668d1435484e974","observation_id":"58dce4d4-e423-4b02-bde8-2f778b39fac8","resolution":{"observed_at":"2026-05-19T14:27:24.261896Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.16883","last_updated":"2026-05-16T08:51:57Z","snapshot_observed_at":"2026-08-03T03:43:27.806285Z","submitted_at":"2026-05-16T08:51:57Z","title":"SE-GA: Memory-Augmented Self-Evolution for GUI Agents","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-19T20:38:37.264654Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.16883"},"observation_digest":"sha256:b08fe0c65f2a7026443e97a33d7eb47405ca7085408232e13f8d8780f351c000","observation_id":"d6bc544c-6a6f-4edf-95b5-827127d58e83","resolution":{"observed_at":"2026-05-19T20:42:46.440909Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.17439","last_updated":"2026-05-19T07:14:09Z","snapshot_observed_at":"2026-07-06T23:28:26.397778Z","submitted_at":"2026-05-17T13:22:22Z","title":"DiagEval: Trajectory-Conditioned Diagnosis for Reliable Software Evaluation with GUI Agents","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-19T23:09:56.130802Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.17439"},"observation_digest":"sha256:7a44f53996035e6d542e05f94db641ff667ab2cd8ecefe78880f0dba71f1f7a4","observation_id":"b4c71a0d-20a2-4e7c-80f4-9c0782067b27","resolution":{"observed_at":"2026-05-19T23:12:51.627186Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.17439","last_updated":"2026-05-19T07:14:09Z","snapshot_observed_at":"2026-07-06T23:28:26.397778Z","submitted_at":"2026-05-17T13:22:22Z","title":"DiagEval: Trajectory-Conditioned Diagnosis for Reliable Software Evaluation with GUI Agents","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-20T13:05:38.485058Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.17439"},"observation_digest":"sha256:9d209b4f59a0c980f07ce8485a9712986971cf10d64dcaf5ec4b9f476765e379","observation_id":"7d1118b9-8b36-4de9-b373-6d3fb5f69a63","resolution":{"observed_at":"2026-05-20T13:08:17.868769Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.18652","last_updated":"2026-05-18T16:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-18T16:57:36Z","title":"MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-20T11:55:36.734758Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.18652"},"observation_digest":"sha256:c1437591769616a965d869072c9363cf1505aa3c6896955fa0587c12a8295847","observation_id":"c756d375-944c-4725-8051-febe9e348588","resolution":{"observed_at":"2026-05-20T11:58:15.052412Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.23939","last_updated":"2026-04-28T11:39:20Z","snapshot_observed_at":"2026-07-31T15:29:41.171388Z","submitted_at":"2026-04-28T11:39:20Z","title":"DRIVE: Modeling Skills at the Reasoning and Interaction Levels for Web Agents under Continual Learning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T09:01:22.412685Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.23939"},"observation_digest":"sha256:e9850906bfb600da036ff1c0b6070e647279e69cfe795b95f5a3e02935fcc80e","observation_id":"2d2e817a-9fb5-4e86-a35b-3a27363b109c","resolution":{"observed_at":"2026-07-01T09:05:36.679252Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2605.30884","last_updated":"2026-05-29T06:17:53Z","snapshot_observed_at":"2026-07-06T23:40:07.833692Z","submitted_at":"2026-05-29T06:17:53Z","title":"GUI-C$^2$: Coarse-to-Fine GUI Grounding via Difficulty-Aware Reinforcement Learning","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-28T22:52:04.755523Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2605.30884"},"observation_digest":"sha256:dca5349543803e4469e55f39764517e27210eb5941c602037a6ab09581d19bb7","observation_id":"4b71b842-1f95-4b7a-83e0-201f7bbc538b","resolution":{"observed_at":"2026-06-28T22:52:44.753598Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2606.00390","last_updated":"2026-05-29T22:12:40Z","snapshot_observed_at":"2026-07-06T23:41:05.637982Z","submitted_at":"2026-05-29T22:12:40Z","title":"Zamba2-VL Technical Report","version":1},"reference_index":130,"source":"pdf_text","source_observed_at":"2026-06-28T22:34:20.970856Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2606.00390"},"observation_digest":"sha256:8c81092b979e078de73887c67941885a0d7a8b15c5ebd68d3672c092348bde38","observation_id":"6bff79e4-8bdd-42fc-b1bc-ea62ac306f02","resolution":{"observed_at":"2026-07-01T19:26:00.692256Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2606.06322","last_updated":"2026-06-04T15:57:29Z","snapshot_observed_at":"2026-08-02T19:13:51.313136Z","submitted_at":"2026-06-04T15:57:29Z","title":"DragOn: A Benchmark and Dataset for Drag-Based GUI Interactions","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-06-28T01:37:45.812577Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2606.06322"},"observation_digest":"sha256:b42e4967b89e0b14716cd354420eb96bc8651ff5bc772421d468be105af85148","observation_id":"65244c57-75c7-4176-b138-98f50716e6b0","resolution":{"observed_at":"2026-06-28T01:41:29.379294Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2606.10522","last_updated":"2026-07-06T06:23:24Z","snapshot_observed_at":"2026-08-02T13:28:24.785118Z","submitted_at":"2026-06-09T07:52:10Z","title":"GUI-AC: Enhancing Continual Learning in GUI Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T13:56:09.049753Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2606.10522"},"observation_digest":"sha256:a3f6611c6308ea32a3e0a07166dac6ac79e884a28fe346d87965da3e7f305320","observation_id":"0a2988c1-d9b4-4695-b0ea-f579bcd5fb7b","resolution":{"observed_at":"2026-07-03T04:27:36.610811Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-12T14:27:05.589465Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents.arXiv preprint arXiv:2410.05243, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.10522","last_updated":"2026-07-06T06:23:24Z","snapshot_observed_at":"2026-08-02T13:28:24.785118Z","submitted_at":"2026-06-09T07:52:10Z","title":"GUI-AC: Enhancing Continual Learning in GUI Agents","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-12T14:27:05.589465Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2606.10522"},"observation_digest":"sha256:9ade29cfd40eccfbc38d9cef7fa425528f81763f9a9d1b783e8fef158dc42b7f","observation_id":"623ba8a8-8461-4193-817c-04ab87c9f2ff","resolution":{"observed_at":"2026-07-12T14:27:05.589465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":"2410.05243","doi":"10.48550/arxiv.2410.05243","metadata_source":"pith","pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","venue":"cs.AI","work_id":"9def1724-6fd2-4d5b-8339-4c1ee76e62f8","year":2024},"citing_paper":{"arxiv_id":"2606.31924","last_updated":"2026-06-30T16:33:03Z","snapshot_observed_at":"2026-08-01T22:17:51.863711Z","submitted_at":"2026-06-30T16:33:03Z","title":"InstanceControl: Controllable Complex Image Generation without Instance Labeling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T05:37:41.030752Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2606.31924"},"observation_digest":"sha256:d2e85b0c5400bf8eb01b51efe4b897efc9c2e7458e52d5244418faa58e190d82","observation_id":"ffd8730a-d5ef-4414-930b-2f7130adf2f0","resolution":{"observed_at":"2026-07-01T10:15:45.069831Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05243","snapshot_observed_at":"2026-07-14T06:25:23.264527Z","title":"Navigating the digital world as humans do: Universal visual grounding for gui agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11185","last_updated":"2026-07-13T07:32:35Z","snapshot_observed_at":"2026-07-16T23:19:14.868685Z","submitted_at":"2026-07-13T07:32:35Z","title":"SCALECUA: Scaling Computer Use Agents with Verifiable Task Synthesis and Efficient Online RL","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-14T06:25:23.264527Z"},"links":{"cited_paper":"/paper/2410.05243","citing_paper":"/paper/2607.11185"},"observation_digest":"sha256:f36794ad9c5924c8353423939431846d0a6a19e2fb5ed919c0db7967c3ec7f78","observation_id":"33f473d4-e0e8-4825-8b6e-65aeb4b337d0","resolution":{"observed_at":"2026-07-14T06:25:23.264527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.05243/citation-record","integrity":"/paper/2410.05243/integrity","json":"/paper/2410.05243/citation-record.json","paper":"/paper/2410.05243"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"click ... then type","venue":null,"work_id":"23c8d22f-9413-422d-bc04-11dfde6dd5dd","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:d081572217f5a5034f49cb7822d972cfbe353d8e9582f0f92250b6032366a6ac","observation_id":"94313550-a39b-40a9-9628-aa13b1abecc5","resolution":{"observed_at":"2026-05-16T21:09:01.412624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We filter out any actions that do not have associated coordinate data, ensuring that only steps with specific visual grounding targets are included in the dataset","venue":null,"work_id":"4a69b9a8-e8ca-4f5d-bf41-eb09660d0534","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:ab98cb799cba53d5ced400a59e5257d50afbc03715383d5f9566c168cb59868d","observation_id":"b41f892f-85ff-4d6f-89d8-479004cc2f94","resolution":{"observed_at":"2026-05-16T21:09:01.417722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"To enhance diversity, two captions per element are randomly selected from the available set of functional captions during data construction","venue":null,"work_id":"aade4fc7-8978-4d0c-9f3f-3708f5f7362e","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:40261f02a7150ec527e30a76d862c28bedf5492a7cdb7cc64c5fef7c16bc9ebf","observation_id":"b7323732-7ed0-49f9-a532-32b0a59745ee","resolution":{"observed_at":"2026-05-16T21:09:01.421529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dc13223e-fa8f-4b7b-a524-d6ff01a0d8f8","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:fd05943c1523d065f8205c6c1b91d4b7f07a236dd7fa476f4de0e0e2e0afb19f","observation_id":"eb14ee81-dd5e-446e-aa73-43d7789fdfca","resolution":{"observed_at":"2026-05-16T21:09:01.424227Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"These annotations contribute to a more diverse set of referring expressions, particularly for action-oriented grounding tasks","venue":null,"work_id":"1c8757f6-dfdc-4a18-a3e3-510b215dae6f","year":2023},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:de170d4239ef14696f12a4a84e713a820ed714c75bd7e5a7b50c596b5ebca6a9","observation_id":"94f6923f-9db9-44bb-9c99-c177b364e63f","resolution":{"observed_at":"2026-05-16T21:09:01.427091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b9148559-c665-407d-afd0-5b333b2083dc","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:041236e1f7f118d7b8d0be96a5f6d58c82c877f28bf2a12060fd8053bb520c43","observation_id":"586ff21f-3a1c-40be-8dd5-6c8ecc02ea2f","resolution":{"observed_at":"2026-05-16T21:09:01.429649Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Due to the huge computation cost of handling high-resolution images, we use LoRA (Hu et al., 2022) for instruction finetuning in the two stages, with a device batch size of 4","venue":null,"work_id":"65bbf84e-7562-4ad5-9b36-22b7e9c5ae85","year":2022},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:117a690cd6aaf5dc9596907829c730b4adc57ddea55bf5f67abd020e9fd08379","observation_id":"3553254c-fd62-4ffb-bf7c-043f0f6ea599","resolution":{"observed_at":"2026-05-16T21:09:01.432448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"38bf2f1b-41f6-4759-bd5a-0f5184ec3cbf","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:1347cd181ab7e8ed5e82337f8d216f179d95eedc221b60f7da4602acfd816375","observation_id":"2c05a570-5f95-4f7a-955d-5965d6da2484","resolution":{"observed_at":"2026-05-16T21:09:01.434840Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a6d5c2a5-1206-4735-a4ac-d2f59dcfad80","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:54f80431e143f0927e4a9ebcda007ffe50adafa91838e50c7d6236e3a47d1a74","observation_id":"0416dc2f-2979-41d4-aea5-99d96a0d33a4","resolution":{"observed_at":"2026-05-16T21:09:01.437120Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"18e479f2-33f5-4468-9054-ca13fc3000a1","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:1fdf986a896375cf5e67a0b9c7a77e19a5b9d7436705f06f8ee81357ab9cdb52","observation_id":"85e2d57b-3942-4dd1-81a3-e9cb11061870","resolution":{"observed_at":"2026-05-16T21:09:01.439984Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"If it is only partially visible, you need to SCROLL DOWN to see the entire element","venue":null,"work_id":"4eff1a6b-35e1-4e7b-a3ac-0ef8148cb6b5","year":null},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:55ee86687fa60fd0b7aeaef4fa194550e50ec8b0307ed9b43d25bd365c393175","observation_id":"c19b5a40-ddd3-48c8-a99c-cd6d8afad2bc","resolution":{"observed_at":"2026-05-16T21:09:01.442604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"the target element","venue":null,"work_id":"886f9613-293c-4ec7-ae03-a63af06ff2b3","year":2025},"citing_paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T21:09:01.397691Z"},"links":{"citing_paper":"/paper/2410.05243"},"observation_digest":"sha256:b751f0b26779a520ffb92272bb2add56bf59aebb3cc9ebaaf46d707cd90cf339","observation_id":"40cba693-0d69-411c-ae79-eb2485df0a42","resolution":{"observed_at":"2026-05-16T21:09:01.445067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2410.05243","last_updated":"2025-06-17T15:06:02Z","latest_version":3,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T19:29:09.714725Z","submitted_at":"2024-10-07T17:47:50Z","title":"Navigating the Digital World as Humans Do: Universal Visual Grounding for GUI Agents"},"reference_resolution":{"displayed":12,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":12},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 12 of 12 outbound references and 45 inbound Pith citation observations for arXiv:2410.05243."}