{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LYWOBDGVZ5H6OOPOAES6JAYFVI","short_pith_number":"pith:LYWOBDGV","schema_version":"1.0","canonical_sha256":"5e2ce08cd5cf4fe739ee0125e48305aa1a3cc26e7c880f13f3e9dbaf0af4f75c","source":{"kind":"arxiv","id":"2505.03570","version":1},"attestation_state":"computed","paper":{"title":"OSUniverse: Benchmark for Multimodal GUI-navigation AI Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Arturo M\\'arquez Flores, Daniel Jeffries, Mariya Davydova, Patrick Barker, Sin\\'ead Ryan","submitted_at":"2025-05-06T14:29:47Z","abstract_excerpt":"In this paper, we introduce OSUniverse: a benchmark of complex, multimodal desktop-oriented tasks for advanced GUI-navigation AI agents that focuses on ease of use, extensibility, comprehensive coverage of test cases, and automated validation. We divide the tasks in increasing levels of complexity, from basic precision clicking to multistep, multiapplication tests requiring dexterity, precision, and clear thinking from the agent. In version one of the benchmark, presented here, we have calibrated the complexity of the benchmark test cases to ensure that the SOTA (State of the Art) agents (at t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.03570","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-06T14:29:47Z","cross_cats_sorted":[],"title_canon_sha256":"f8b422722a4b99666af646d810da58e70efb50037e69e01429df16ca8238cd0f","abstract_canon_sha256":"256c810b3bf4902a595373c3b6721e3aaefe8740da232c6e5c704d8e65cd550a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:18.450435Z","signature_b64":"1mx32c2QMn3MF6+fY5iUasG3lTn/AxaqSfhJHjo61afXzmqlVlF17z7+t0HupelHeayz6oV9xS2aqIyht+L9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e2ce08cd5cf4fe739ee0125e48305aa1a3cc26e7c880f13f3e9dbaf0af4f75c","last_reissued_at":"2026-07-05T10:59:18.449960Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:18.449960Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OSUniverse: Benchmark for Multimodal GUI-navigation AI Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Arturo M\\'arquez Flores, Daniel Jeffries, Mariya Davydova, Patrick Barker, Sin\\'ead Ryan","submitted_at":"2025-05-06T14:29:47Z","abstract_excerpt":"In this paper, we introduce OSUniverse: a benchmark of complex, multimodal desktop-oriented tasks for advanced GUI-navigation AI agents that focuses on ease of use, extensibility, comprehensive coverage of test cases, and automated validation. We divide the tasks in increasing levels of complexity, from basic precision clicking to multistep, multiapplication tests requiring dexterity, precision, and clear thinking from the agent. In version one of the benchmark, presented here, we have calibrated the complexity of the benchmark test cases to ensure that the SOTA (State of the Art) agents (at t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.03570","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.03570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.03570","created_at":"2026-07-05T10:59:18.450017+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.03570v1","created_at":"2026-07-05T10:59:18.450017+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.03570","created_at":"2026-07-05T10:59:18.450017+00:00"},{"alias_kind":"pith_short_12","alias_value":"LYWOBDGVZ5H6","created_at":"2026-07-05T10:59:18.450017+00:00"},{"alias_kind":"pith_short_16","alias_value":"LYWOBDGVZ5H6OOPO","created_at":"2026-07-05T10:59:18.450017+00:00"},{"alias_kind":"pith_short_8","alias_value":"LYWOBDGV","created_at":"2026-07-05T10:59:18.450017+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28480","citing_title":"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29705","citing_title":"GUICrafter: Weakly-Supervised GUI Agent Leveraging Massive Unannotated Screenshots","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18842","citing_title":"GUIGuard-Bench: Toward a General Evaluation for Privacy-Preserving GUI Agents","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27776","citing_title":"WindowsWorld: A Process-Centric Benchmark of Autonomous GUI Agents in Professional Cross-Application Environments","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI","json":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI.json","graph_json":"https://pith.science/api/pith-number/LYWOBDGVZ5H6OOPOAES6JAYFVI/graph.json","events_json":"https://pith.science/api/pith-number/LYWOBDGVZ5H6OOPOAES6JAYFVI/events.json","paper":"https://pith.science/paper/LYWOBDGV"},"agent_actions":{"view_html":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI","download_json":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI.json","view_paper":"https://pith.science/paper/LYWOBDGV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.03570&json=true","fetch_graph":"https://pith.science/api/pith-number/LYWOBDGVZ5H6OOPOAES6JAYFVI/graph.json","fetch_events":"https://pith.science/api/pith-number/LYWOBDGVZ5H6OOPOAES6JAYFVI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI/action/storage_attestation","attest_author":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI/action/author_attestation","sign_citation":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI/action/citation_signature","submit_replication":"https://pith.science/pith/LYWOBDGVZ5H6OOPOAES6JAYFVI/action/replication_record"}},"created_at":"2026-07-05T10:59:18.450017+00:00","updated_at":"2026-07-05T10:59:18.450017+00:00"}