{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LQAVFH32YWOAIHG4KTUOO4EAWH","short_pith_number":"pith:LQAVFH32","schema_version":"1.0","canonical_sha256":"5c01529f7ac59c041cdc54e8e77080b1ea0632cadb09fb8a3927844ca38727cd","source":{"kind":"arxiv","id":"2506.01616","version":1},"attestation_state":"computed","paper":{"title":"MLA-Trust: Benchmarking Trustworthiness of Multimodal LLM Agents in GUI Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hang Su, Jiawei Chen, Jun Luo, Jun Zhu, Xiao Yang, Yinpeng Dong, Zhengwei Fang","submitted_at":"2025-06-02T12:56:27Z","abstract_excerpt":"The emergence of multimodal LLM-based agents (MLAs) has transformed interaction paradigms by seamlessly integrating vision, language, action and dynamic environments, enabling unprecedented autonomous capabilities across GUI applications ranging from web automation to mobile systems. However, MLAs introduce critical trustworthiness challenges that extend far beyond traditional language models' limitations, as they can directly modify digital states and trigger irreversible real-world consequences. Existing benchmarks inadequately tackle these unique challenges posed by MLAs' actionable outputs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.01616","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-02T12:56:27Z","cross_cats_sorted":[],"title_canon_sha256":"fc61b33bef1b59a14af3be0cee57d3e77adbd7fa13d3ad1e606cfc2bc562bb2d","abstract_canon_sha256":"f729af2c699e3b080e55f349e97108899dddf106f495f4376155d1e36307b315"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:16.873851Z","signature_b64":"iJwdmPr5e0hrPfUpEOAr0L4Fz30Q7t9fAIhsSvodAlnnPjErXTJkP7Ntqkr86S7LdDIJbaaMF2KUCDD0mlYaAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5c01529f7ac59c041cdc54e8e77080b1ea0632cadb09fb8a3927844ca38727cd","last_reissued_at":"2026-07-05T11:14:16.873398Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:16.873398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLA-Trust: Benchmarking Trustworthiness of Multimodal LLM Agents in GUI Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Hang Su, Jiawei Chen, Jun Luo, Jun Zhu, Xiao Yang, Yinpeng Dong, Zhengwei Fang","submitted_at":"2025-06-02T12:56:27Z","abstract_excerpt":"The emergence of multimodal LLM-based agents (MLAs) has transformed interaction paradigms by seamlessly integrating vision, language, action and dynamic environments, enabling unprecedented autonomous capabilities across GUI applications ranging from web automation to mobile systems. However, MLAs introduce critical trustworthiness challenges that extend far beyond traditional language models' limitations, as they can directly modify digital states and trigger irreversible real-world consequences. Existing benchmarks inadequately tackle these unique challenges posed by MLAs' actionable outputs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.01616","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.01616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.01616","created_at":"2026-07-05T11:14:16.873457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.01616v1","created_at":"2026-07-05T11:14:16.873457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.01616","created_at":"2026-07-05T11:14:16.873457+00:00"},{"alias_kind":"pith_short_12","alias_value":"LQAVFH32YWOA","created_at":"2026-07-05T11:14:16.873457+00:00"},{"alias_kind":"pith_short_16","alias_value":"LQAVFH32YWOAIHG4","created_at":"2026-07-05T11:14:16.873457+00:00"},{"alias_kind":"pith_short_8","alias_value":"LQAVFH32","created_at":"2026-07-05T11:14:16.873457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12666","citing_title":"CAPED: Context-Aware Privacy Exposure Defense for Mobile GUI Agents","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23989","citing_title":"Towards trustworthy agentic AI: a comprehensive survey of safety, robustness, privacy, and system security","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20874","citing_title":"Governance by Construction for Generalist Agents","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12382","citing_title":"Exploring the Secondary Risks of Large Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07553","citing_title":"VeriOS: Query-Driven Proactive Human-Agent-GUI Interaction for Trustworthy OS Agents","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2512.00412","citing_title":"Red Teaming Large Reasoning Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18842","citing_title":"GUIGuard-Bench: Toward a General Evaluation for Privacy-Preserving GUI Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24348","citing_title":"OS-SPEAR: A Toolkit for the Safety, Performance,Efficiency, and Robustness Analysis of OS Agents","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH","json":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH.json","graph_json":"https://pith.science/api/pith-number/LQAVFH32YWOAIHG4KTUOO4EAWH/graph.json","events_json":"https://pith.science/api/pith-number/LQAVFH32YWOAIHG4KTUOO4EAWH/events.json","paper":"https://pith.science/paper/LQAVFH32"},"agent_actions":{"view_html":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH","download_json":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH.json","view_paper":"https://pith.science/paper/LQAVFH32","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.01616&json=true","fetch_graph":"https://pith.science/api/pith-number/LQAVFH32YWOAIHG4KTUOO4EAWH/graph.json","fetch_events":"https://pith.science/api/pith-number/LQAVFH32YWOAIHG4KTUOO4EAWH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH/action/storage_attestation","attest_author":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH/action/author_attestation","sign_citation":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH/action/citation_signature","submit_replication":"https://pith.science/pith/LQAVFH32YWOAIHG4KTUOO4EAWH/action/replication_record"}},"created_at":"2026-07-05T11:14:16.873457+00:00","updated_at":"2026-07-05T11:14:16.873457+00:00"}