{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5WLNJWNQGMVEBZXVAN7SOXXOCJ","short_pith_number":"pith:5WLNJWNQ","schema_version":"1.0","canonical_sha256":"ed96d4d9b0332a40e6f5037f275eee124560f3a662fe631c7c0760cb106d402e","source":{"kind":"arxiv","id":"2406.10819","version":2},"attestation_state":"computed","paper":{"title":"GUI-World: A Video Benchmark and Dataset for Multimodal GUI-oriented Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chenlong Wang, Chujie Gao, Dongping Chen, Huichi Zhou, Jianfeng Gao, Jingyu Tang, Lichao Sun, Liuyi Chen, Pan Zhou, Qihui Zhang, Siyuan Wu, Tianshuo Zhou, Yao Wan, Yi Gui, Yilin Bai, Yiqiang Li, Yue Huang, Yue Yu, Zhen Li, Zhigang He","submitted_at":"2024-06-16T06:56:53Z","abstract_excerpt":"Recently, Multimodal Large Language Models (MLLMs) have been used as agents to control keyboard and mouse inputs by directly perceiving the Graphical User Interface (GUI) and generating corresponding commands. However, current agents primarily demonstrate strong understanding capabilities in static environments and are mainly applied to relatively simple domains, such as Web or mobile interfaces. We argue that a robust GUI agent should be capable of perceiving temporal information on the GUI, including dynamic Web content and multi-step tasks. Additionally, it should possess a comprehensive un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10819","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-16T06:56:53Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"b1a7e50c574f204f44a08ea94424d1a27f6ef40ef589fd414ff7f959c8b31657","abstract_canon_sha256":"acd3119d8a435a1d4ef8a8ad155ce978bae69424cf9204c214e15ad0c41605fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:27.428256Z","signature_b64":"YUnBM6lK+wfQQnBVJsdkV9qrUg5IB8rdtMQdNU0UB2QWjV44uda2NmeddghRZ8lCr9RqXXuZ4HAk1eBwX98uBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed96d4d9b0332a40e6f5037f275eee124560f3a662fe631c7c0760cb106d402e","last_reissued_at":"2026-07-05T10:37:27.427339Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:27.427339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GUI-World: A Video Benchmark and Dataset for Multimodal GUI-oriented Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chenlong Wang, Chujie Gao, Dongping Chen, Huichi Zhou, Jianfeng Gao, Jingyu Tang, Lichao Sun, Liuyi Chen, Pan Zhou, Qihui Zhang, Siyuan Wu, Tianshuo Zhou, Yao Wan, Yi Gui, Yilin Bai, Yiqiang Li, Yue Huang, Yue Yu, Zhen Li, Zhigang He","submitted_at":"2024-06-16T06:56:53Z","abstract_excerpt":"Recently, Multimodal Large Language Models (MLLMs) have been used as agents to control keyboard and mouse inputs by directly perceiving the Graphical User Interface (GUI) and generating corresponding commands. However, current agents primarily demonstrate strong understanding capabilities in static environments and are mainly applied to relatively simple domains, such as Web or mobile interfaces. We argue that a robust GUI agent should be capable of perceiving temporal information on the GUI, including dynamic Web content and multi-step tasks. Additionally, it should possess a comprehensive un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10819","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10819/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10819","created_at":"2026-07-05T10:37:27.427457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10819v2","created_at":"2026-07-05T10:37:27.427457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10819","created_at":"2026-07-05T10:37:27.427457+00:00"},{"alias_kind":"pith_short_12","alias_value":"5WLNJWNQGMVE","created_at":"2026-07-05T10:37:27.427457+00:00"},{"alias_kind":"pith_short_16","alias_value":"5WLNJWNQGMVEBZXV","created_at":"2026-07-05T10:37:27.427457+00:00"},{"alias_kind":"pith_short_8","alias_value":"5WLNJWNQ","created_at":"2026-07-05T10:37:27.427457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19741","citing_title":"CityRAG: Stepping Into a City via Spatially-Grounded Video Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02522","citing_title":"Moment-Video: Diagnosing Temporal Fidelity of Video MLLMs on Momentary Visual Events","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29472","citing_title":"Agent-Computer Observation Interfaces Enable Dynamic Computer Use","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29705","citing_title":"GUICrafter: Weakly-Supervised GUI Agent Leveraging Massive Unannotated Screenshots","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18758","citing_title":"OmniGUI: Benchmarking GUI Agents in Omni-Modal Smartphone Environments","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18048","citing_title":"DocOS: Towards Proactive Document-Guided Actions in GUI Agents","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2512.00756","citing_title":"MPR-GUI: Benchmarking and Enhancing Multilingual Perception and Reasoning in GUI Agents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01785","citing_title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26148","citing_title":"Beyond Screenshots: Evaluating VLMs' Understanding of UI Animations","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19742","citing_title":"PlayCoder: Making LLM-Generated GUI Code Playable","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18543","citing_title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ","json":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ.json","graph_json":"https://pith.science/api/pith-number/5WLNJWNQGMVEBZXVAN7SOXXOCJ/graph.json","events_json":"https://pith.science/api/pith-number/5WLNJWNQGMVEBZXVAN7SOXXOCJ/events.json","paper":"https://pith.science/paper/5WLNJWNQ"},"agent_actions":{"view_html":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ","download_json":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ.json","view_paper":"https://pith.science/paper/5WLNJWNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10819&json=true","fetch_graph":"https://pith.science/api/pith-number/5WLNJWNQGMVEBZXVAN7SOXXOCJ/graph.json","fetch_events":"https://pith.science/api/pith-number/5WLNJWNQGMVEBZXVAN7SOXXOCJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ/action/storage_attestation","attest_author":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ/action/author_attestation","sign_citation":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ/action/citation_signature","submit_replication":"https://pith.science/pith/5WLNJWNQGMVEBZXVAN7SOXXOCJ/action/replication_record"}},"created_at":"2026-07-05T10:37:27.427457+00:00","updated_at":"2026-07-05T10:37:27.427457+00:00"}