{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UAME32WQFUZPIEZHKVWK4TZFUK","short_pith_number":"pith:UAME32WQ","schema_version":"1.0","canonical_sha256":"a0184dead02d32f41327556cae4f25a293f9d077980091f9ddf8060680bfd7a7","source":{"kind":"arxiv","id":"2503.15661","version":2},"attestation_state":"computed","paper":{"title":"UI-Vision: A Desktop-centric GUI Benchmark for Visual Perception and Interaction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Aishwarya Agrawal, Christopher Pal, David Vazquez, Juan A. Rodriguez, Kevin Qinghong Lin, Montek Kalsi, M. Tamer \\\"Ozsu, Nicolas Chapados, Perouz Taslakian, Rabiul Awal, Sai Rajeswar, Shravan Nayak, Spandana Gella, Xiangru Jian","submitted_at":"2025-03-19T19:26:17Z","abstract_excerpt":"Autonomous agents that navigate Graphical User Interfaces (GUIs) to automate tasks like document editing and file management can greatly enhance computer workflows. While existing research focuses on online settings, desktop environments, critical for many professional and everyday tasks, remain underexplored due to data collection challenges and licensing issues. We introduce UI-Vision, the first comprehensive, license-permissive benchmark for offline, fine-grained evaluation of computer use agents in real-world desktop environments. Unlike online benchmarks, UI-Vision provides: (i) dense, hi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.15661","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-19T19:26:17Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5a4d5ea4ac4419a4ae0a15f023bea3fc9a6f240a37fc77191787135843a466e4","abstract_canon_sha256":"0a5c58c20e21973defb61eac2b26b2914d50fd4140102e30b4e40bf5c2c7365f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:08.326417Z","signature_b64":"RyQj6dC0s4QGiWP1WPzkqPo9GJ97rWb9EFJ5BO3UZHfWv6lpO3XFMaxaIsbuyiHpc5rgANW4zSDzB2Srx+edCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0184dead02d32f41327556cae4f25a293f9d077980091f9ddf8060680bfd7a7","last_reissued_at":"2026-07-05T10:59:08.325884Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:08.325884Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UI-Vision: A Desktop-centric GUI Benchmark for Visual Perception and Interaction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Aishwarya Agrawal, Christopher Pal, David Vazquez, Juan A. Rodriguez, Kevin Qinghong Lin, Montek Kalsi, M. Tamer \\\"Ozsu, Nicolas Chapados, Perouz Taslakian, Rabiul Awal, Sai Rajeswar, Shravan Nayak, Spandana Gella, Xiangru Jian","submitted_at":"2025-03-19T19:26:17Z","abstract_excerpt":"Autonomous agents that navigate Graphical User Interfaces (GUIs) to automate tasks like document editing and file management can greatly enhance computer workflows. While existing research focuses on online settings, desktop environments, critical for many professional and everyday tasks, remain underexplored due to data collection challenges and licensing issues. We introduce UI-Vision, the first comprehensive, license-permissive benchmark for offline, fine-grained evaluation of computer use agents in real-world desktop environments. Unlike online benchmarks, UI-Vision provides: (i) dense, hi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.15661","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.15661/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.15661","created_at":"2026-07-05T10:59:08.325949+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.15661v2","created_at":"2026-07-05T10:59:08.325949+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.15661","created_at":"2026-07-05T10:59:08.325949+00:00"},{"alias_kind":"pith_short_12","alias_value":"UAME32WQFUZP","created_at":"2026-07-05T10:59:08.325949+00:00"},{"alias_kind":"pith_short_16","alias_value":"UAME32WQFUZPIEZH","created_at":"2026-07-05T10:59:08.325949+00:00"},{"alias_kind":"pith_short_8","alias_value":"UAME32WQ","created_at":"2026-07-05T10:59:08.325949+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25760","citing_title":"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30084","citing_title":"One Forward Beats Two: InnerZoom for Accurate and Efficient GUI Grounding","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25571","citing_title":"AnE: Pushing the Reasoning Frontier of Multimodal LLMs via Anchor Evolution","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19260","citing_title":"AQuaUI: Visual Token Reduction for GUI Agents with Adaptive Quadtrees","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12549","citing_title":"What Happens Before Decoding? Prefill Determines GUI Grounding in VLMs","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12501","citing_title":"Covering Human Action Space for Computer Use: Data Synthesis and Benchmark","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27955","citing_title":"GUI Agents with Reinforcement Learning: Toward Digital Inhabitants","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00642","citing_title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00642","citing_title":"Learn where to Click from Yourself: On-Policy Self-Distillation for GUI Grounding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21268","citing_title":"Measure Twice, Click Once: Co-evolving Proposer and Visual Critic via Reinforcement Learning for GUI Grounding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07413","citing_title":"FORGE: Fine-grained Multimodal Evaluation for Manufacturing Scenarios","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14113","citing_title":"UI-Zoomer: Uncertainty-Driven Adaptive Zoom-In for GUI Grounding","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK","json":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK.json","graph_json":"https://pith.science/api/pith-number/UAME32WQFUZPIEZHKVWK4TZFUK/graph.json","events_json":"https://pith.science/api/pith-number/UAME32WQFUZPIEZHKVWK4TZFUK/events.json","paper":"https://pith.science/paper/UAME32WQ"},"agent_actions":{"view_html":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK","download_json":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK.json","view_paper":"https://pith.science/paper/UAME32WQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.15661&json=true","fetch_graph":"https://pith.science/api/pith-number/UAME32WQFUZPIEZHKVWK4TZFUK/graph.json","fetch_events":"https://pith.science/api/pith-number/UAME32WQFUZPIEZHKVWK4TZFUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK/action/storage_attestation","attest_author":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK/action/author_attestation","sign_citation":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK/action/citation_signature","submit_replication":"https://pith.science/pith/UAME32WQFUZPIEZHKVWK4TZFUK/action/replication_record"}},"created_at":"2026-07-05T10:59:08.325949+00:00","updated_at":"2026-07-05T10:59:08.325949+00:00"}