{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NICX4UOAGKBUJ6SJBBZSMAFYFF","short_pith_number":"pith:NICX4UOA","schema_version":"1.0","canonical_sha256":"6a057e51c0328344fa4908732600b8296aa5e1c0bf9aae829fae08885cd26380","source":{"kind":"arxiv","id":"2406.08451","version":2},"attestation_state":"computed","paper":{"title":"GUIOdyssey: A Comprehensive Dataset for Cross-App GUI Navigation on Mobile Devices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Botong Chen, Boxuan Li, Fanqing Meng, Kaipeng Zhang, Lingxiao Du, Ping Luo, Quanfeng Lu, Siyuan Huang, Wenqi Shao, Zitao Liu","submitted_at":"2024-06-12T17:44:26Z","abstract_excerpt":"Autonomous Graphical User Interface (GUI) navigation agents can enhance user experience in communication, entertainment, and productivity by streamlining workflows and reducing manual intervention. However, prior GUI agents often trained with datasets comprising tasks that can be completed within a single app, leading to poor performance in cross-app navigation. To address this problem, we present GUIOdyssey, a comprehensive dataset for cross-app mobile GUI navigation. GUIOdyssey comprises 8,334 episodes with an average of 15.3 steps per episode, covering 6 mobile devices, 212 distinct apps, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08451","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-12T17:44:26Z","cross_cats_sorted":[],"title_canon_sha256":"4d94e7693de65d29e8852f16c8d353a718537712ca69490701c34402ee41f8ad","abstract_canon_sha256":"a839336786cdc6851c5b3c4787cb1c6790677028338c2ee1a2f38a560221d778"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:34.717904Z","signature_b64":"nMljt+WTXjYo7+7ggfruDASBlIqQb7GZDI0xWB1wjAFN8qUdcumFwV057FvmCRzfMn+SDqfvdeqOs9vv9qcHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a057e51c0328344fa4908732600b8296aa5e1c0bf9aae829fae08885cd26380","last_reissued_at":"2026-07-05T11:46:34.717434Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:34.717434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GUIOdyssey: A Comprehensive Dataset for Cross-App GUI Navigation on Mobile Devices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Botong Chen, Boxuan Li, Fanqing Meng, Kaipeng Zhang, Lingxiao Du, Ping Luo, Quanfeng Lu, Siyuan Huang, Wenqi Shao, Zitao Liu","submitted_at":"2024-06-12T17:44:26Z","abstract_excerpt":"Autonomous Graphical User Interface (GUI) navigation agents can enhance user experience in communication, entertainment, and productivity by streamlining workflows and reducing manual intervention. However, prior GUI agents often trained with datasets comprising tasks that can be completed within a single app, leading to poor performance in cross-app navigation. To address this problem, we present GUIOdyssey, a comprehensive dataset for cross-app mobile GUI navigation. GUIOdyssey comprises 8,334 episodes with an average of 15.3 steps per episode, covering 6 mobile devices, 212 distinct apps, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08451","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08451/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08451","created_at":"2026-07-05T11:46:34.717498+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08451v2","created_at":"2026-07-05T11:46:34.717498+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08451","created_at":"2026-07-05T11:46:34.717498+00:00"},{"alias_kind":"pith_short_12","alias_value":"NICX4UOAGKBU","created_at":"2026-07-05T11:46:34.717498+00:00"},{"alias_kind":"pith_short_16","alias_value":"NICX4UOAGKBUJ6SJ","created_at":"2026-07-05T11:46:34.717498+00:00"},{"alias_kind":"pith_short_8","alias_value":"NICX4UOA","created_at":"2026-07-05T11:46:34.717498+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31365","citing_title":"Learning to Adapt: Self-Improving Web Agent via Cognitive-Aware Exploration","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17656","citing_title":"MUIAnno: An Expert-Annotated Dataset and Evaluation Benchmark for Mobile UI Understanding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16883","citing_title":"SE-GA: Memory-Augmented Self-Evolution for GUI Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.20332","citing_title":"Mobile-R1: Towards Interactive Capability for VLM-Based Mobile Agent via Systematic Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07553","citing_title":"VeriOS: Query-Driven Proactive Human-Agent-GUI Interaction for Trustworthy OS Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04454","citing_title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02345","citing_title":"UI-Oceanus: Scaling GUI Agents with Synthetic Environmental Dynamics","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2602.21858","citing_title":"ProactiveMobile: A Comprehensive Benchmark for Boosting Proactive Intelligence on Mobile Devices","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05044","citing_title":"WebFactory: Automated Compression of Foundational Language Intelligence into Grounded Web Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10458","citing_title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2410.23218","citing_title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09442","citing_title":"UIPress: Bringing Optical Token Compression to UI-to-Code Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":168,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF","json":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF.json","graph_json":"https://pith.science/api/pith-number/NICX4UOAGKBUJ6SJBBZSMAFYFF/graph.json","events_json":"https://pith.science/api/pith-number/NICX4UOAGKBUJ6SJBBZSMAFYFF/events.json","paper":"https://pith.science/paper/NICX4UOA"},"agent_actions":{"view_html":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF","download_json":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF.json","view_paper":"https://pith.science/paper/NICX4UOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08451&json=true","fetch_graph":"https://pith.science/api/pith-number/NICX4UOAGKBUJ6SJBBZSMAFYFF/graph.json","fetch_events":"https://pith.science/api/pith-number/NICX4UOAGKBUJ6SJBBZSMAFYFF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF/action/storage_attestation","attest_author":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF/action/author_attestation","sign_citation":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF/action/citation_signature","submit_replication":"https://pith.science/pith/NICX4UOAGKBUJ6SJBBZSMAFYFF/action/replication_record"}},"created_at":"2026-07-05T11:46:34.717498+00:00","updated_at":"2026-07-05T11:46:34.717498+00:00"}