{"as_of":"2026-08-14T06:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b238106e3d2fb34093aba155cb79ba76f12495129491737c87343e84551166d8","coverage":[{"denominator":16,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T02:00:17.765986Z","state":"measured"},{"denominator":16,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":16,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.07630/citation-record","integrity":"/paper/2605.07630/integrity","json":"/paper/2605.07630/citation-record.json","paper":"/paper/2605.07630"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Agentharm: A benchmark for measuring harmfulness of llm agents","venue":null,"work_id":"085b251f-8686-4739-acdc-4f5659c21065","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:f4a6271483b5975512a54763eb913e9675999e4310d2ad1507e4eba85bee3cec","observation_id":"ce3c76f2-9ee6-4060-9e74-e211edd11f7d","resolution":{"observed_at":"2026-05-14T14:46:27.426460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.20333","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T18:25:57.693276Z","title":"GhostEI-Bench: Do Mobile Agents Resilience to Environmental Injection in Dynamic On-Device Environments?","venue":null,"work_id":"b334713b-2584-421a-a1fe-3085ab1ca188","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:6a6fb682eaa733261ab3cb56da88e0996e67c728f77d1193d7f9cccb0db2d6b5","observation_id":"abf75e8b-cc76-422f-bdc3-592bbf827a98","resolution":{"observed_at":"2026-05-11T04:00:56.770633Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00820","last_updated":"2024-10-28T17:05:10Z","snapshot_observed_at":"2026-08-13T05:17:20.876741Z","submitted_at":"2024-10-28T17:05:10Z","title":"AutoGLM: Autonomous Foundation Agents for GUIs","version":1},"cited_work":{"arxiv_id":"2411.00820","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.00820","snapshot_observed_at":"2026-07-03T04:27:36.601401Z","title":"L., Sun, J., Wang, J., et al","venue":null,"work_id":"fb970438-4def-4de7-8a40-9ee7c2236fcd","year":2024},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2411.00820","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:210775bf20fa05ee93acb190f2f423e5affc0b76de934b7fb8bd68b54b849ea8","observation_id":"ad6516a5-bed4-4210-b651-10f98571fd45","resolution":{"observed_at":"2026-05-11T04:00:56.793948Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.12349","last_updated":"2026-07-15T15:32:11Z","snapshot_observed_at":"2026-08-09T21:35:22.237317Z","submitted_at":"2026-01-18T10:54:54Z","title":"Mind the Gap: Action Rebinding Attacks against Android GUI Agents","version":3},"cited_work":{"arxiv_id":"2601.12349","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2601.12349","snapshot_observed_at":"2026-07-16T02:22:32.277203Z","title":"Zero-permission manipulation: Can we trust large multimodal model powered gui agents?","venue":null,"work_id":"a81dd481-8f05-421a-ae0d-6d4119d26d65","year":2026},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2601.12349","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:afda37156102ca77d1074966c5a2e710b863fa2129c1e0227401767c646554d9","observation_id":"377dfc56-2591-42af-8622-822a491aaf02","resolution":{"observed_at":"2026-07-16T02:22:32.277203Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12326","last_updated":"2025-01-21T17:48:10Z","snapshot_observed_at":"2026-07-06T20:23:58.426780Z","submitted_at":"2025-01-21T17:48:10Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","version":1},"cited_work":{"arxiv_id":"2501.12326","doi":"10.48550/arxiv.2501.12326","metadata_source":"pith","pith_arxiv_id":"2501.12326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","venue":"cs.AI","work_id":"0bbcf263-a46d-4525-a438-11fce3316568","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2501.12326","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:537412a2941a02652424beadc3b2fd38335c7caee5bd90ef0c03f8058e92cc8b","observation_id":"c7fde410-27de-46a1-ab6e-21b3bf5b8e4e","resolution":{"observed_at":"2026-05-11T04:00:56.786560Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:19.638951+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:19.638951+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14573","last_updated":"2025-04-06T20:37:50Z","snapshot_observed_at":"2026-08-13T23:11:04.935073Z","submitted_at":"2024-05-23T13:48:54Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","version":5},"cited_work":{"arxiv_id":"2405.14573","doi":"10.48550/arxiv.2405.14573","metadata_source":"pith","pith_arxiv_id":"2405.14573","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","venue":"cs.AI","work_id":"c5116d19-d3d3-40fd-9620-f7489812a9ba","year":2024},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2405.14573","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:a05d77e9b084057a35c691e6f8ea28b034a2240dac63c2fcceb55b336f443dde","observation_id":"2a4d5afe-d468-4959-b21a-20f23f214a92","resolution":{"observed_at":"2026-05-13T12:06:13.928391Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.23434","doi":"10.48550/arxiv.2503.23434","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards trustworthy gui agents: A survey","venue":"arXiv (Cornell University)","work_id":"e5e673ed-c68f-4401-ad43-0d70b7677c6d","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:28e707b442bb1925eae3ab58251d89da8c00e8ae5bfeff2d39216452d4dca182","observation_id":"37bb027b-27af-44a9-836e-f985a2854440","resolution":{"observed_at":"2026-05-11T04:00:56.757685Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:73f8332401a7ecf0b27c72a6056b51bcf0874302dc1cb472c80e273c4c6bc03f","observation_id":"1f19ccb3-0b86-49bb-9aaf-b658398ff782","resolution":{"observed_at":"2026-05-11T04:00:56.726372Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:6b216321595d4e1aca93ecfaada33d205f4d943652dcdebe34e3d7aa7e239a6e","observation_id":"23a95f3a-a187-40aa-902d-3ce753d5764a","resolution":{"observed_at":"2026-05-11T02:00:51.755223Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.06134","doi":"10.48550/arxiv.2507.06134","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vijayvargiya, A","venue":"ArXiv.org","work_id":"7315530a-8de5-452c-9c59-6e0cc7436bf1","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:1a30f21b691914877f828f68e11bc5f9336dd4012e610e55a77b14994ef92c5e","observation_id":"b7a65a44-4fed-4cf5-8098-243b4f69e5b4","resolution":{"observed_at":"2026-05-11T04:00:56.777244Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.18842","last_updated":"2026-05-13T07:11:36Z","snapshot_observed_at":"2026-08-11T18:25:16.311048Z","submitted_at":"2026-01-26T11:33:40Z","title":"GUIGuard-Bench: Toward a General Evaluation for Privacy-Preserving GUI Agents","version":3},"cited_work":{"arxiv_id":"2601.18842","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.18842","snapshot_observed_at":"2026-07-03T12:28:07.462121Z","title":"GUIGuard: Toward a General Framework for Privacy-Preserving GUI Agents, jan 2026","venue":"cs.CR","work_id":"3d5b9bb9-826f-4510-a0b8-f4320bc9f953","year":2026},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2601.18842","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:6bc1af25c585e5fb8d63706817b730c8590cdb2f8c2bddc60db2343a182199d7","observation_id":"5bb2d03f-7f1a-4ceb-8250-7c1d7f5d1e78","resolution":{"observed_at":"2026-05-14T01:55:13.551984Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":"44e22d4a-d80c-4f29-a363-7e89e9471224","year":2024},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:d7afe4d0080de1fb6153365dd81e59c109e40833a36f8b62e07e0ec9f0c89fc4","observation_id":"17b374d4-b844-4cb5-acd9-0db7011ae089","resolution":{"observed_at":"2026-05-14T14:46:27.424667Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.16855","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:59:46.401274Z","title":"Mobile-agent-v3","venue":null,"work_id":"48f0aefc-ae32-479a-809e-be6bc0441fd3","year":2026},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:32847ca3d0a37426b2a9abafb990957157e8497889eb8c635ed23c7b4e3d2af8","observation_id":"f8130c6f-be8a-44b2-9160-cfc8570ab04e","resolution":{"observed_at":"2026-05-11T04:00:56.752873Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.00618","last_updated":"2025-06-20T01:24:06Z","snapshot_observed_at":"2026-08-14T03:16:20.343826Z","submitted_at":"2025-05-31T16:04:59Z","title":"RiOSWorld: Benchmarking the Risk of Multimodal Computer-Use Agents","version":3},"cited_work":{"arxiv_id":"2506.00618","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.00618","snapshot_observed_at":"2026-07-01T10:15:44.749818Z","title":"Riosworld: Benchmarking the risk of multimodal computer-use agents","venue":null,"work_id":"26d40ed4-6126-4f39-a622-abccf8432eaa","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"cited_paper":"/paper/2506.00618","citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:e495e930fd970baa8b64a1f76423b416f0c8583f7bff1b3df139d8ac8f871a61","observation_id":"2bd2be58-e9cb-47bf-9b50-bbe51e238bf6","resolution":{"observed_at":"2026-05-11T04:00:56.737761Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.22047","doi":"10.48550/arxiv.2512.22047","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mai-ui technical report: Real-world centric foundation gui agents","venue":"arXiv (Cornell University)","work_id":"97605953-1cbe-4ff9-b1e4-339feb4e776a","year":2025},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:10887c8b7d5e9a7352a58180ca8debb642a1187a9c2fc4d8344428601e096a90","observation_id":"1189d69f-3a2e-4472-9e12-159061c6fc0c","resolution":{"observed_at":"2026-05-11T04:00:56.721714Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Benchmarks such as MobileSafetyBench Lee et al","venue":null,"work_id":"ba51fe7a-ee3a-4917-b1dd-4c509c3d7108","year":2026},"citing_paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T02:00:17.765986Z"},"links":{"citing_paper":"/paper/2605.07630"},"observation_digest":"sha256:8f3e692ca72f3381b057112deacf92e8887264f8470c080339c1ad515efdb05a","observation_id":"ffc256a1-ecda-4e5f-bab2-d7aa90a6562d","resolution":{"observed_at":"2026-05-14T14:46:27.428484Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.07630","last_updated":"2026-05-08T11:58:57Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-11T10:59:29.868795Z","submitted_at":"2026-05-08T11:58:57Z","title":"Safe, or Simply Incapable? Rethinking Safety Evaluation for Phone-Use Agents"},"reference_resolution":{"displayed":16,"state_counts":{"malformed_identifier":2,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":0,"verified_exact":10,"verified_fuzzy":2},"total_outbound_references":16},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 16 of 16 outbound references and 0 inbound Pith citation observations for arXiv:2605.07630."}