{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:ENCFUWTBLGU5RL2EQ7GGSCIDOR","short_pith_number":"pith:ENCFUWTB","schema_version":"1.0","canonical_sha256":"23445a5a6159a9d8af4487cc69090374691a0fc9522f14eaeb83e8dafb8bb83d","source":{"kind":"arxiv","id":"2608.10692","version":1},"attestation_state":"computed","paper":{"title":"SPIEval: Evaluating Large Language Models as Mobile Assistants over Scattered Personal Information","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dingwei Zhu, Junjie Ye, Ming Zhang, Pluto Zhou, Qi Zhang, Shaofan Liu, Shihan Dou, Tao Gui, Weichao Wang, Wenjie Fu, Xin Zhao, Xuanjing Huang, Yujiong Shen, Yulun Zhu, Zhuohui Sheng","submitted_at":"2026-08-11T09:14:53Z","abstract_excerpt":"Large language models (LLMs) are increasingly deployed as mobile assistants, where a key challenge is leveraging personal information scattered across multiple applications (apps) to complete user instructions. However, due to the lack of dedicated benchmarks, their capabilities remain poorly understood. To address this gap, we introduce SPIEval, a human-curated benchmark grounded in five cognitive capabilities (i.e., reasoning, disambiguation, integration, preference inference, and multi-intent decomposition). SPIEval comprises 250 tasks spanning 4,335 personal records distributed across 10 a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.10692","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-11T09:14:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6316ff95bcf2bcd9e65c8b59d919c5c3bbc2a02a19e189abb7753ffb76adf659","abstract_canon_sha256":"a276bbc7de623db1cdde0c26659a5cc8cb8c543dce731069d9b46e90aa145195"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-12T01:23:04.297560Z","signature_b64":"ZbBePTJwA8k6ym8mZ9I8oFx/SBi0hVwRVUzmvlOW6lmcPrdsokTtWmkBjXKrF+p3I8SNONRYOnnpsDNlLqeuAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23445a5a6159a9d8af4487cc69090374691a0fc9522f14eaeb83e8dafb8bb83d","last_reissued_at":"2026-08-12T01:23:04.295536Z","signature_status":"signed_v1","first_computed_at":"2026-08-12T01:23:04.295536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SPIEval: Evaluating Large Language Models as Mobile Assistants over Scattered Personal Information","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dingwei Zhu, Junjie Ye, Ming Zhang, Pluto Zhou, Qi Zhang, Shaofan Liu, Shihan Dou, Tao Gui, Weichao Wang, Wenjie Fu, Xin Zhao, Xuanjing Huang, Yujiong Shen, Yulun Zhu, Zhuohui Sheng","submitted_at":"2026-08-11T09:14:53Z","abstract_excerpt":"Large language models (LLMs) are increasingly deployed as mobile assistants, where a key challenge is leveraging personal information scattered across multiple applications (apps) to complete user instructions. However, due to the lack of dedicated benchmarks, their capabilities remain poorly understood. To address this gap, we introduce SPIEval, a human-curated benchmark grounded in five cognitive capabilities (i.e., reasoning, disambiguation, integration, preference inference, and multi-intent decomposition). SPIEval comprises 250 tasks spanning 4,335 personal records distributed across 10 a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.10692","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.10692/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.10692","created_at":"2026-08-12T01:23:04.299656+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.10692v1","created_at":"2026-08-12T01:23:04.299656+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.10692","created_at":"2026-08-12T01:23:04.299656+00:00"},{"alias_kind":"pith_short_12","alias_value":"ENCFUWTBLGU5","created_at":"2026-08-12T01:23:04.299656+00:00"},{"alias_kind":"pith_short_16","alias_value":"ENCFUWTBLGU5RL2E","created_at":"2026-08-12T01:23:04.299656+00:00"},{"alias_kind":"pith_short_8","alias_value":"ENCFUWTB","created_at":"2026-08-12T01:23:04.299656+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR","json":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR.json","graph_json":"https://pith.science/api/pith-number/ENCFUWTBLGU5RL2EQ7GGSCIDOR/graph.json","events_json":"https://pith.science/api/pith-number/ENCFUWTBLGU5RL2EQ7GGSCIDOR/events.json","paper":"https://pith.science/paper/ENCFUWTB"},"agent_actions":{"view_html":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR","download_json":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR.json","view_paper":"https://pith.science/paper/ENCFUWTB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.10692&json=true","fetch_graph":"https://pith.science/api/pith-number/ENCFUWTBLGU5RL2EQ7GGSCIDOR/graph.json","fetch_events":"https://pith.science/api/pith-number/ENCFUWTBLGU5RL2EQ7GGSCIDOR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR/action/storage_attestation","attest_author":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR/action/author_attestation","sign_citation":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR/action/citation_signature","submit_replication":"https://pith.science/pith/ENCFUWTBLGU5RL2EQ7GGSCIDOR/action/replication_record"}},"created_at":"2026-08-12T01:23:04.299656+00:00","updated_at":"2026-08-12T01:23:04.299656+00:00"}