{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X3HGNIAVBHGU4ACXRDADVLNVB4","short_pith_number":"pith:X3HGNIAV","schema_version":"1.0","canonical_sha256":"bece66a01509cd4e005788c03aadb50f2e8836d9202b3b161b7862479ebf795d","source":{"kind":"arxiv","id":"2406.08184","version":1},"attestation_state":"computed","paper":{"title":"MobileAgentBench: An Efficient and User-Friendly Benchmark for Mobile LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.AI","authors_text":"Guodong Mao, Luyuan Wang, Qinmin Wang, Shoufa Chen, Tianchen Min, Wei Chen, Yiwei Zha, Yongyu Deng","submitted_at":"2024-06-12T13:14:50Z","abstract_excerpt":"Large language model (LLM)-based mobile agents are increasingly popular due to their capability to interact directly with mobile phone Graphic User Interfaces (GUIs) and their potential to autonomously manage daily tasks. Despite their promising prospects in both academic and industrial sectors, little research has focused on benchmarking the performance of existing mobile agents, due to the inexhaustible states of apps and the vague definition of feasible action sequences. To address this challenge, we propose an efficient and user-friendly benchmark, MobileAgentBench, designed to alleviate t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08184","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-12T13:14:50Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"a9f28728ad99a20486ad66540dafa16f3ff11b9980591f632949a7c0d4bd3b37","abstract_canon_sha256":"87cd82f8f940216f82086dce061d6485ee97c8c7aaa71254c92743ce9ed47e9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:56.280017Z","signature_b64":"yh1OEoBllkx14lpILVl5aV9V3cP3HdBrX/GOsF2r5+M6+BJ94kZZ1paTMv/QGvtXjor+tDIGCgYiSmWWNAJDCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bece66a01509cd4e005788c03aadb50f2e8836d9202b3b161b7862479ebf795d","last_reissued_at":"2026-07-05T08:30:56.279513Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:56.279513Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MobileAgentBench: An Efficient and User-Friendly Benchmark for Mobile LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.AI","authors_text":"Guodong Mao, Luyuan Wang, Qinmin Wang, Shoufa Chen, Tianchen Min, Wei Chen, Yiwei Zha, Yongyu Deng","submitted_at":"2024-06-12T13:14:50Z","abstract_excerpt":"Large language model (LLM)-based mobile agents are increasingly popular due to their capability to interact directly with mobile phone Graphic User Interfaces (GUIs) and their potential to autonomously manage daily tasks. Despite their promising prospects in both academic and industrial sectors, little research has focused on benchmarking the performance of existing mobile agents, due to the inexhaustible states of apps and the vague definition of feasible action sequences. To address this challenge, we propose an efficient and user-friendly benchmark, MobileAgentBench, designed to alleviate t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08184","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08184/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08184","created_at":"2026-07-05T08:30:56.279574+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08184v1","created_at":"2026-07-05T08:30:56.279574+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08184","created_at":"2026-07-05T08:30:56.279574+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3HGNIAVBHGU","created_at":"2026-07-05T08:30:56.279574+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3HGNIAVBHGU4ACX","created_at":"2026-07-05T08:30:56.279574+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3HGNIAV","created_at":"2026-07-05T08:30:56.279574+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23049","citing_title":"PhoneBuddy: Training Open Models for Agentic Phone Use","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18048","citing_title":"DocOS: Towards Proactive Document-Guided Actions in GUI Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18048","citing_title":"DocOS: Towards Proactive Document-Guided Actions in GUI Agents","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17406","citing_title":"Rethinking Side-Channel Analysis: Automated Discovery and Analysis of Side-Channel Leakage with LLM-Assisted Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06477","citing_title":"MAS-Bench: A Unified Benchmark for Shortcut-Augmented Hybrid Mobile GUI Agents","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10371","citing_title":"AgentProg: Empowering Long-Horizon GUI Agents with Program-Guided Context Management","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12634","citing_title":"MobiBench: Multi-Branch, Modular Benchmark for Mobile GUI Agents","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11763","citing_title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13564","citing_title":"Memory in the Age of AI Agents","ref_index":291,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07767","citing_title":"Administrative Decentralization in Edge-Cloud Multi-Agent for Mobile Automation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17817","citing_title":"Do LLMs Need to See Everything? A Benchmark and Study of Failures in LLM-driven Smartphone Automation using Screentext vs. Screenshots","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4","json":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4.json","graph_json":"https://pith.science/api/pith-number/X3HGNIAVBHGU4ACXRDADVLNVB4/graph.json","events_json":"https://pith.science/api/pith-number/X3HGNIAVBHGU4ACXRDADVLNVB4/events.json","paper":"https://pith.science/paper/X3HGNIAV"},"agent_actions":{"view_html":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4","download_json":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4.json","view_paper":"https://pith.science/paper/X3HGNIAV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08184&json=true","fetch_graph":"https://pith.science/api/pith-number/X3HGNIAVBHGU4ACXRDADVLNVB4/graph.json","fetch_events":"https://pith.science/api/pith-number/X3HGNIAVBHGU4ACXRDADVLNVB4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4/action/storage_attestation","attest_author":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4/action/author_attestation","sign_citation":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4/action/citation_signature","submit_replication":"https://pith.science/pith/X3HGNIAVBHGU4ACXRDADVLNVB4/action/replication_record"}},"created_at":"2026-07-05T08:30:56.279574+00:00","updated_at":"2026-07-05T08:30:56.279574+00:00"}