{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XKENEIE6CKARXURQFSSL3JJ7D6","short_pith_number":"pith:XKENEIE6","schema_version":"1.0","canonical_sha256":"ba88d2209e12811bd2302ca4bda53f1fa19851f6bdb6aba3117f2b618c0d08b6","source":{"kind":"arxiv","id":"2305.08144","version":4},"attestation_state":"computed","paper":{"title":"Mobile-Env: Building Qualified Evaluation Benchmarks for LLM-GUI Interaction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Danyang Zhang, Hongshen Xu, Kai Yu, Lu Chen, Ruisheng Cao, Rui Xie, Situo Zhang, Siyuan Chen, Tianbao Xie, Zhennan Shen, Zihan Zhao","submitted_at":"2023-05-14T12:31:03Z","abstract_excerpt":"The Graphical User Interface (GUI) is pivotal for human interaction with the digital world, enabling efficient device control and the completion of complex tasks. Recent progress in Large Language Models (LLMs) and Vision Language Models (VLMs) offers the chance to create advanced GUI agents. To ensure their effectiveness, there's a pressing need for qualified benchmarks that provide trustworthy and reproducible evaluations -- a challenge current benchmarks often fail to address. To tackle this issue, we introduce Mobile-Env, a comprehensive toolkit tailored for creating GUI benchmarks in the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.08144","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-05-14T12:31:03Z","cross_cats_sorted":[],"title_canon_sha256":"7be1cd5f796e5d9d297d676e7ebdc594c526fe70191b64e46483463933c74bae","abstract_canon_sha256":"0b4653d48e5c96edd583665d614e45c03566012a76ae4ea7deb88cae5b6aa212"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:03.607467Z","signature_b64":"ldKUfKu4z7FG3ToVouTzXqOaZ+ng+WdMT8v5quAPn9k2PfqJBqz2ipgkQDRzs1HSnmNRi81KnV/NYGb2r8rxAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba88d2209e12811bd2302ca4bda53f1fa19851f6bdb6aba3117f2b618c0d08b6","last_reissued_at":"2026-07-05T08:31:03.606986Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:03.606986Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mobile-Env: Building Qualified Evaluation Benchmarks for LLM-GUI Interaction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Danyang Zhang, Hongshen Xu, Kai Yu, Lu Chen, Ruisheng Cao, Rui Xie, Situo Zhang, Siyuan Chen, Tianbao Xie, Zhennan Shen, Zihan Zhao","submitted_at":"2023-05-14T12:31:03Z","abstract_excerpt":"The Graphical User Interface (GUI) is pivotal for human interaction with the digital world, enabling efficient device control and the completion of complex tasks. Recent progress in Large Language Models (LLMs) and Vision Language Models (VLMs) offers the chance to create advanced GUI agents. To ensure their effectiveness, there's a pressing need for qualified benchmarks that provide trustworthy and reproducible evaluations -- a challenge current benchmarks often fail to address. To tackle this issue, we introduce Mobile-Env, a comprehensive toolkit tailored for creating GUI benchmarks in the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.08144","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.08144/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.08144","created_at":"2026-07-05T08:31:03.607046+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.08144v4","created_at":"2026-07-05T08:31:03.607046+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.08144","created_at":"2026-07-05T08:31:03.607046+00:00"},{"alias_kind":"pith_short_12","alias_value":"XKENEIE6CKAR","created_at":"2026-07-05T08:31:03.607046+00:00"},{"alias_kind":"pith_short_16","alias_value":"XKENEIE6CKARXURQ","created_at":"2026-07-05T08:31:03.607046+00:00"},{"alias_kind":"pith_short_8","alias_value":"XKENEIE6","created_at":"2026-07-05T08:31:03.607046+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09764","citing_title":"iOSWorld: A Benchmark for Personally Intelligent Phone Agents","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17829","citing_title":"Interactive Evaluation Requires a Design Science","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19679","citing_title":"InquireMobile: Teaching VLM-based Mobile Agent to Request Human Assistance via Reinforcement Fine-Tuning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05459","citing_title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12634","citing_title":"MobiBench: Multi-Branch, Modular Benchmark for Mobile GUI Agents","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09514","citing_title":"EcoGym: Evaluating LLMs for Long-Horizon Plan-and-Execute in Interactive Economies","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2308.11432","citing_title":"A Survey on Large Language Model based Autonomous Agents","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07972","citing_title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6","json":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6.json","graph_json":"https://pith.science/api/pith-number/XKENEIE6CKARXURQFSSL3JJ7D6/graph.json","events_json":"https://pith.science/api/pith-number/XKENEIE6CKARXURQFSSL3JJ7D6/events.json","paper":"https://pith.science/paper/XKENEIE6"},"agent_actions":{"view_html":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6","download_json":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6.json","view_paper":"https://pith.science/paper/XKENEIE6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.08144&json=true","fetch_graph":"https://pith.science/api/pith-number/XKENEIE6CKARXURQFSSL3JJ7D6/graph.json","fetch_events":"https://pith.science/api/pith-number/XKENEIE6CKARXURQFSSL3JJ7D6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6/action/storage_attestation","attest_author":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6/action/author_attestation","sign_citation":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6/action/citation_signature","submit_replication":"https://pith.science/pith/XKENEIE6CKARXURQFSSL3JJ7D6/action/replication_record"}},"created_at":"2026-07-05T08:31:03.607046+00:00","updated_at":"2026-07-05T08:31:03.607046+00:00"}