{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:R6CLSEYLKRE42CS5DSUOSDLKML","short_pith_number":"pith:R6CLSEYL","schema_version":"1.0","canonical_sha256":"8f84b9130b5449cd0a5d1ca8e90d6a62ebb16ba6889dca4931aec7ba054c33bc","source":{"kind":"arxiv","id":"2504.07981","version":1},"attestation_state":"computed","paper":{"title":"ScreenSpot-Pro: GUI Grounding for Professional High-Resolution Computer Use","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC","cs.MM"],"primary_cat":"cs.CV","authors_text":"Hongzhan Lin, Jing Ma, Kaixin Li, Tat-Seng Chua, Yuchen Tian, Zhiyong Huang, Ziyang Luo, Ziyang Meng","submitted_at":"2025-04-04T14:25:17Z","abstract_excerpt":"Recent advancements in Multi-modal Large Language Models (MLLMs) have led to significant progress in developing GUI agents for general tasks such as web browsing and mobile phone use. However, their application in professional domains remains under-explored. These specialized workflows introduce unique challenges for GUI perception models, including high-resolution displays, smaller target sizes, and complex environments. In this paper, we introduce ScreenSpot-Pro, a new benchmark designed to rigorously evaluate the grounding capabilities of MLLMs in high-resolution professional settings. The "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07981","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-04T14:25:17Z","cross_cats_sorted":["cs.HC","cs.MM"],"title_canon_sha256":"299fe652aac65e7ca26a97f7c12a19da9ce71f154653b6a90da70b08e5ca29f3","abstract_canon_sha256":"bc2bae133e7d4ae4de044b0fef42a03454ccbe6401a8ac08cac34ec59da20165"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:40.481039Z","signature_b64":"XKavhpP6BvpXdVBvuoJikT08A6+iQ36jSb+NXxATCnS4iMh7T3QRKQYf8UXtsx409tMibXR5ax/8HiMorAXXAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f84b9130b5449cd0a5d1ca8e90d6a62ebb16ba6889dca4931aec7ba054c33bc","last_reissued_at":"2026-07-05T10:47:40.480567Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:40.480567Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ScreenSpot-Pro: GUI Grounding for Professional High-Resolution Computer Use","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC","cs.MM"],"primary_cat":"cs.CV","authors_text":"Hongzhan Lin, Jing Ma, Kaixin Li, Tat-Seng Chua, Yuchen Tian, Zhiyong Huang, Ziyang Luo, Ziyang Meng","submitted_at":"2025-04-04T14:25:17Z","abstract_excerpt":"Recent advancements in Multi-modal Large Language Models (MLLMs) have led to significant progress in developing GUI agents for general tasks such as web browsing and mobile phone use. However, their application in professional domains remains under-explored. These specialized workflows introduce unique challenges for GUI perception models, including high-resolution displays, smaller target sizes, and complex environments. In this paper, we introduce ScreenSpot-Pro, a new benchmark designed to rigorously evaluate the grounding capabilities of MLLMs in high-resolution professional settings. The "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07981","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07981","created_at":"2026-07-05T10:47:40.480623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07981v1","created_at":"2026-07-05T10:47:40.480623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07981","created_at":"2026-07-05T10:47:40.480623+00:00"},{"alias_kind":"pith_short_12","alias_value":"R6CLSEYLKRE4","created_at":"2026-07-05T10:47:40.480623+00:00"},{"alias_kind":"pith_short_16","alias_value":"R6CLSEYLKRE42CS5","created_at":"2026-07-05T10:47:40.480623+00:00"},{"alias_kind":"pith_short_8","alias_value":"R6CLSEYL","created_at":"2026-07-05T10:47:40.480623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25760","citing_title":"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31270","citing_title":"Learning from Failure: Inference-Time Self-Improvement for Computer-Use Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29537","citing_title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29705","citing_title":"GUICrafter: Weakly-Supervised GUI Agent Leveraging Massive Unannotated Screenshots","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30884","citing_title":"GUI-C$^2$: Coarse-to-Fine GUI Grounding via Difficulty-Aware Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06322","citing_title":"DragOn: A Benchmark and Dataset for Drag-Based GUI Interactions","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10887","citing_title":"InfantAgent-Next: A Multimodal Generalist Agent for Automated Computer Interaction","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17656","citing_title":"MUIAnno: An Expert-Annotated Dataset and Evaluation Benchmark for Mobile UI Understanding","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15542","citing_title":"DRS-GUI: Dynamic Region Search for Training-Free GUI Grounding","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10610","citing_title":"LaSM: Layer-wise Scaling Mechanism for Defending Pop-up Attack on GUI Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19679","citing_title":"InquireMobile: Teaching VLM-based Mobile Agent to Request Human Assistance via Reinforcement Fine-Tuning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07553","citing_title":"VeriOS: Query-Driven Proactive Human-Agent-GUI Interaction for Trustworthy OS Agents","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21982","citing_title":"RISK: A Framework for GUI Agents in E-commerce Risk Management","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22102","citing_title":"Mitigating Coordinate Prediction Bias from Positional Encoding Failures","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2510.24168","citing_title":"MGA: Memory-Driven GUI Agent for Observation-Centric Interaction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05791","citing_title":"GTA1: GUI Test-time Scaling Agent","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19219","citing_title":"Selective LoRA for Visual Tokens and Attention Heads","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21858","citing_title":"ProactiveMobile: A Comprehensive Benchmark for Boosting Proactive Intelligence on Mobile Devices","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06664","citing_title":"BAMI: Training-Free Bias Mitigation in GUI Grounding","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13019","citing_title":"PrecisionCUA: Iterative Visual Refinement for Pixel-Precise Cursor Grounding in Code Editors","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14262","citing_title":"GUI-Perturbed: Domain Randomization Reveals Systematic Brittleness in GUI Grounding Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15376","citing_title":"Zoom Consistency: A Free Confidence Signal in Multi-Step Visual Grounding Pipelines","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML","json":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML.json","graph_json":"https://pith.science/api/pith-number/R6CLSEYLKRE42CS5DSUOSDLKML/graph.json","events_json":"https://pith.science/api/pith-number/R6CLSEYLKRE42CS5DSUOSDLKML/events.json","paper":"https://pith.science/paper/R6CLSEYL"},"agent_actions":{"view_html":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML","download_json":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML.json","view_paper":"https://pith.science/paper/R6CLSEYL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07981&json=true","fetch_graph":"https://pith.science/api/pith-number/R6CLSEYLKRE42CS5DSUOSDLKML/graph.json","fetch_events":"https://pith.science/api/pith-number/R6CLSEYLKRE42CS5DSUOSDLKML/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML/action/storage_attestation","attest_author":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML/action/author_attestation","sign_citation":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML/action/citation_signature","submit_replication":"https://pith.science/pith/R6CLSEYLKRE42CS5DSUOSDLKML/action/replication_record"}},"created_at":"2026-07-05T10:47:40.480623+00:00","updated_at":"2026-07-05T10:47:40.480623+00:00"}