{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FAZDVGYQM3RSAIEESFJDHAB5KQ","short_pith_number":"pith:FAZDVGYQ","schema_version":"1.0","canonical_sha256":"28323a9b1066e3202084915233803d5422fd012e59b69a1bec4261608b323964","source":{"kind":"arxiv","id":"2406.19263","version":2},"attestation_state":"computed","paper":{"title":"Read Anywhere Pointed: Layout-aware GUI Screen Reading with Tree-of-Lens Grounding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Ching-Chen Kuo, Jie Yang, Lei Ding, Shan Jiang, Xin Eric Wang, Xinze Guan, Yang Zhao, Yi Zhang, Yue Fan","submitted_at":"2024-06-27T15:34:16Z","abstract_excerpt":"Graphical User Interfaces (GUIs) are central to our interaction with digital devices and growing efforts have been made to build models for various GUI understanding tasks. However, these efforts largely overlook an important GUI-referring task: screen reading based on user-indicated points, which we name the Screen Point-and-Read (ScreenPR) task. Currently, this task is predominantly handled by rigid accessible screen reading tools, in great need of new models driven by advancements in Multimodal Large Language Models (MLLMs). In this paper, we propose a Tree-of-Lens (ToL) agent, utilizing a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19263","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-27T15:34:16Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"ed8e478c616b03031c3ee6a629f123809d5390d5db65c9b95c92cba518b317d4","abstract_canon_sha256":"a1c31601e7b99d5ee9cdf00abc1448aa6ccf75689e0096237ca8d3e8662609bb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:26:28.711734Z","signature_b64":"iFzQ/M1piBVbqBW7LVURQy11UzXcK2lWuA16nHN1/WgFi4B3Qh67Ffc5HdhA22hjsPMhTBbu5fXwxM330TJIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28323a9b1066e3202084915233803d5422fd012e59b69a1bec4261608b323964","last_reissued_at":"2026-07-05T09:26:28.711265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:26:28.711265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Read Anywhere Pointed: Layout-aware GUI Screen Reading with Tree-of-Lens Grounding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Ching-Chen Kuo, Jie Yang, Lei Ding, Shan Jiang, Xin Eric Wang, Xinze Guan, Yang Zhao, Yi Zhang, Yue Fan","submitted_at":"2024-06-27T15:34:16Z","abstract_excerpt":"Graphical User Interfaces (GUIs) are central to our interaction with digital devices and growing efforts have been made to build models for various GUI understanding tasks. However, these efforts largely overlook an important GUI-referring task: screen reading based on user-indicated points, which we name the Screen Point-and-Read (ScreenPR) task. Currently, this task is predominantly handled by rigid accessible screen reading tools, in great need of new models driven by advancements in Multimodal Large Language Models (MLLMs). In this paper, we propose a Tree-of-Lens (ToL) agent, utilizing a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19263","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19263/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19263","created_at":"2026-07-05T09:26:28.711324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19263v2","created_at":"2026-07-05T09:26:28.711324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19263","created_at":"2026-07-05T09:26:28.711324+00:00"},{"alias_kind":"pith_short_12","alias_value":"FAZDVGYQM3RS","created_at":"2026-07-05T09:26:28.711324+00:00"},{"alias_kind":"pith_short_16","alias_value":"FAZDVGYQM3RSAIEE","created_at":"2026-07-05T09:26:28.711324+00:00"},{"alias_kind":"pith_short_8","alias_value":"FAZDVGYQ","created_at":"2026-07-05T09:26:28.711324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08226","citing_title":"TRISHUL: Towards Region Identification and Screen Hierarchy Understanding for Large VLM based GUI Agents","ref_index":2023,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ","json":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ.json","graph_json":"https://pith.science/api/pith-number/FAZDVGYQM3RSAIEESFJDHAB5KQ/graph.json","events_json":"https://pith.science/api/pith-number/FAZDVGYQM3RSAIEESFJDHAB5KQ/events.json","paper":"https://pith.science/paper/FAZDVGYQ"},"agent_actions":{"view_html":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ","download_json":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ.json","view_paper":"https://pith.science/paper/FAZDVGYQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19263&json=true","fetch_graph":"https://pith.science/api/pith-number/FAZDVGYQM3RSAIEESFJDHAB5KQ/graph.json","fetch_events":"https://pith.science/api/pith-number/FAZDVGYQM3RSAIEESFJDHAB5KQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ/action/storage_attestation","attest_author":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ/action/author_attestation","sign_citation":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ/action/citation_signature","submit_replication":"https://pith.science/pith/FAZDVGYQM3RSAIEESFJDHAB5KQ/action/replication_record"}},"created_at":"2026-07-05T09:26:28.711324+00:00","updated_at":"2026-07-05T09:26:28.711324+00:00"}