{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5SBL4BL4WOB2JVZVQQ3LLXONNZ","short_pith_number":"pith:5SBL4BL4","schema_version":"1.0","canonical_sha256":"ec82be057cb383a4d7358436b5ddcd6e68b8890272a5ea2667f519298d2fd099","source":{"kind":"arxiv","id":"2505.00684","version":2},"attestation_state":"computed","paper":{"title":"Visual Test-time Scaling for GUI Agent Grounding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Honglak Lee, Justin Johnson, Lajanugen Logeswaran, Tiange Luo","submitted_at":"2025-05-01T17:45:59Z","abstract_excerpt":"We introduce RegionFocus, a visual test-time scaling approach for Vision Language Model Agents. Understanding webpages is challenging due to the visual complexity of GUI images and the large number of interface elements, making accurate action selection difficult. Our approach dynamically zooms in on relevant regions, reducing background clutter and improving grounding accuracy. To support this process, we propose an image-as-map mechanism that visualizes key landmarks at each step, providing a transparent action record and enables the agent to effectively choose among action candidates. Even "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00684","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-01T17:45:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"76d6707d763cf5996fdbe603d28510cc8c0722659166e06d50580c87f0f37de6","abstract_canon_sha256":"134017e47fce4b99ae63e13a0a0e6cc136c660c5ecb6b05c15f2a929150799d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:31.373654Z","signature_b64":"c0G0JUfpHqLPqB/6iSHVniJAfs56WRZISQygPhCd0NOhwr9wbZSIHsPBbtGx7AJ/DySOKC+UmnPyg75dLkvoDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec82be057cb383a4d7358436b5ddcd6e68b8890272a5ea2667f519298d2fd099","last_reissued_at":"2026-07-05T11:36:31.373099Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:31.373099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Test-time Scaling for GUI Agent Grounding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Honglak Lee, Justin Johnson, Lajanugen Logeswaran, Tiange Luo","submitted_at":"2025-05-01T17:45:59Z","abstract_excerpt":"We introduce RegionFocus, a visual test-time scaling approach for Vision Language Model Agents. Understanding webpages is challenging due to the visual complexity of GUI images and the large number of interface elements, making accurate action selection difficult. Our approach dynamically zooms in on relevant regions, reducing background clutter and improving grounding accuracy. To support this process, we propose an image-as-map mechanism that visualizes key landmarks at each step, providing a transparent action record and enables the agent to effectively choose among action candidates. Even "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00684","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00684","created_at":"2026-07-05T11:36:31.373162+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00684v2","created_at":"2026-07-05T11:36:31.373162+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00684","created_at":"2026-07-05T11:36:31.373162+00:00"},{"alias_kind":"pith_short_12","alias_value":"5SBL4BL4WOB2","created_at":"2026-07-05T11:36:31.373162+00:00"},{"alias_kind":"pith_short_16","alias_value":"5SBL4BL4WOB2JVZV","created_at":"2026-07-05T11:36:31.373162+00:00"},{"alias_kind":"pith_short_8","alias_value":"5SBL4BL4","created_at":"2026-07-05T11:36:31.373162+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04046","citing_title":"Dive into the Scene: Breaking the Perceptual Bottleneck in Vision-Language Decision Making via Focus Plan Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21268","citing_title":"Measure Twice, Click Once: Co-evolving Proposer and Visual Critic via Reinforcement Learning for GUI Grounding","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02630","citing_title":"AutoFocus: Uncertainty-Aware Active Visual Search for GUI Grounding","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ","json":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ.json","graph_json":"https://pith.science/api/pith-number/5SBL4BL4WOB2JVZVQQ3LLXONNZ/graph.json","events_json":"https://pith.science/api/pith-number/5SBL4BL4WOB2JVZVQQ3LLXONNZ/events.json","paper":"https://pith.science/paper/5SBL4BL4"},"agent_actions":{"view_html":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ","download_json":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ.json","view_paper":"https://pith.science/paper/5SBL4BL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00684&json=true","fetch_graph":"https://pith.science/api/pith-number/5SBL4BL4WOB2JVZVQQ3LLXONNZ/graph.json","fetch_events":"https://pith.science/api/pith-number/5SBL4BL4WOB2JVZVQQ3LLXONNZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ/action/storage_attestation","attest_author":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ/action/author_attestation","sign_citation":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ/action/citation_signature","submit_replication":"https://pith.science/pith/5SBL4BL4WOB2JVZVQQ3LLXONNZ/action/replication_record"}},"created_at":"2026-07-05T11:36:31.373162+00:00","updated_at":"2026-07-05T11:36:31.373162+00:00"}