{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F4UAYZG2NRHFOQGR6DADEPICMM","short_pith_number":"pith:F4UAYZG2","schema_version":"1.0","canonical_sha256":"2f280c64da6c4e5740d1f0c0323d02633c8fcddb2978153280e95cd1873f7178","source":{"kind":"arxiv","id":"2410.15164","version":3},"attestation_state":"computed","paper":{"title":"SPA-Bench: A Comprehensive Benchmark for SmartPhone Agent Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bin Xie, Derek Yuen, Gongwei Chen, Jianye Hao, Jingxuan Chen, Jun Wang, Kaiwen Zhou, Kun Shao, Liqiang Nie, Li Yixing, Rui Shao, Shuai Wang, Weiwen Liu, Xurui Zhou, Yasheng Wang, Yuhao Yang, Zhihao Wu","submitted_at":"2024-10-19T17:28:48Z","abstract_excerpt":"Smartphone agents are increasingly important for helping users control devices efficiently, with (Multimodal) Large Language Model (MLLM)-based approaches emerging as key contenders. Fairly comparing these agents is essential but challenging, requiring a varied task scope, the integration of agents with different implementations, and a generalisable evaluation pipeline to assess their strengths and weaknesses. In this paper, we present SPA-Bench, a comprehensive SmartPhone Agent Benchmark designed to evaluate (M)LLM-based agents in an interactive environment that simulates real-world condition"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15164","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-19T17:28:48Z","cross_cats_sorted":[],"title_canon_sha256":"0b73c7f7d4f63b36753fb468130fa8735acc02e27ed95ad63dfa48e9520cd35a","abstract_canon_sha256":"e6ed275c3ae10371aea718bd77b21c1de70419caec2e2c3e59952a21dfe8e2d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:42:13.108859Z","signature_b64":"GRXzpatMS6kLhDqkh9mAYlJmZOh3EuK1D0yQO/nX9i7xuOrVvbo0GL65hnql3Tx64CUSKFiW2CtBQy4p+oYaCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f280c64da6c4e5740d1f0c0323d02633c8fcddb2978153280e95cd1873f7178","last_reissued_at":"2026-07-05T10:42:13.108291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:42:13.108291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SPA-Bench: A Comprehensive Benchmark for SmartPhone Agent Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bin Xie, Derek Yuen, Gongwei Chen, Jianye Hao, Jingxuan Chen, Jun Wang, Kaiwen Zhou, Kun Shao, Liqiang Nie, Li Yixing, Rui Shao, Shuai Wang, Weiwen Liu, Xurui Zhou, Yasheng Wang, Yuhao Yang, Zhihao Wu","submitted_at":"2024-10-19T17:28:48Z","abstract_excerpt":"Smartphone agents are increasingly important for helping users control devices efficiently, with (Multimodal) Large Language Model (MLLM)-based approaches emerging as key contenders. Fairly comparing these agents is essential but challenging, requiring a varied task scope, the integration of agents with different implementations, and a generalisable evaluation pipeline to assess their strengths and weaknesses. In this paper, we present SPA-Bench, a comprehensive SmartPhone Agent Benchmark designed to evaluate (M)LLM-based agents in an interactive environment that simulates real-world condition"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15164","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15164/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15164","created_at":"2026-07-05T10:42:13.108357+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15164v3","created_at":"2026-07-05T10:42:13.108357+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15164","created_at":"2026-07-05T10:42:13.108357+00:00"},{"alias_kind":"pith_short_12","alias_value":"F4UAYZG2NRHF","created_at":"2026-07-05T10:42:13.108357+00:00"},{"alias_kind":"pith_short_16","alias_value":"F4UAYZG2NRHFOQGR","created_at":"2026-07-05T10:42:13.108357+00:00"},{"alias_kind":"pith_short_8","alias_value":"F4UAYZG2","created_at":"2026-07-05T10:42:13.108357+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM","json":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM.json","graph_json":"https://pith.science/api/pith-number/F4UAYZG2NRHFOQGR6DADEPICMM/graph.json","events_json":"https://pith.science/api/pith-number/F4UAYZG2NRHFOQGR6DADEPICMM/events.json","paper":"https://pith.science/paper/F4UAYZG2"},"agent_actions":{"view_html":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM","download_json":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM.json","view_paper":"https://pith.science/paper/F4UAYZG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15164&json=true","fetch_graph":"https://pith.science/api/pith-number/F4UAYZG2NRHFOQGR6DADEPICMM/graph.json","fetch_events":"https://pith.science/api/pith-number/F4UAYZG2NRHFOQGR6DADEPICMM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM/action/storage_attestation","attest_author":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM/action/author_attestation","sign_citation":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM/action/citation_signature","submit_replication":"https://pith.science/pith/F4UAYZG2NRHFOQGR6DADEPICMM/action/replication_record"}},"created_at":"2026-07-05T10:42:13.108357+00:00","updated_at":"2026-07-05T10:42:13.108357+00:00"}