{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SLYYAOPB7T4EGGMZPM4SSXDPKB","short_pith_number":"pith:SLYYAOPB","schema_version":"1.0","canonical_sha256":"92f18039e1fcf84319997b39295c6f505a88c8415499567023c2255e09a2f995","source":{"kind":"arxiv","id":"2402.17553","version":3},"attestation_state":"computed","paper":{"title":"OmniACT: A Dataset and Benchmark for Enabling Multimodal Generalist Autonomous Agents for Desktop and Web","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.HC"],"primary_cat":"cs.AI","authors_text":"Jing Yu Koh, Kiran Kamble, Melisa Russak, Raghav Kapoor, Ruslan Salakhutdinov, Waseem Alshikh, Yash Parag Butala","submitted_at":"2024-02-27T14:47:53Z","abstract_excerpt":"For decades, human-computer interaction has fundamentally been manual. Even today, almost all productive work done on the computer necessitates human input at every step. Autonomous virtual agents represent an exciting step in automating many of these menial tasks. Virtual agents would empower users with limited technical proficiency to harness the full possibilities of computer systems. They could also enable the efficient streamlining of numerous computer tasks, ranging from calendar management to complex travel bookings, with minimal human intervention. In this paper, we introduce OmniACT, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17553","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-02-27T14:47:53Z","cross_cats_sorted":["cs.CL","cs.CV","cs.HC"],"title_canon_sha256":"39233992480a5c79425316deaba31c07fc4c262da8449e47ed578c712d064c10","abstract_canon_sha256":"6f162764f6cd95b2a95269e84687cf1cd93a4a1d232a493510ccb662e4307743"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:22.051832Z","signature_b64":"2D3FE978zQGM0A4gGTjM1ihtCR1YNgXo1+Q1OH4tmmYq4qkZLmsCForquzjpBDhRjpEjIMLM960B0kTdj9c0Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92f18039e1fcf84319997b39295c6f505a88c8415499567023c2255e09a2f995","last_reissued_at":"2026-07-05T08:46:22.051423Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:22.051423Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OmniACT: A Dataset and Benchmark for Enabling Multimodal Generalist Autonomous Agents for Desktop and Web","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.HC"],"primary_cat":"cs.AI","authors_text":"Jing Yu Koh, Kiran Kamble, Melisa Russak, Raghav Kapoor, Ruslan Salakhutdinov, Waseem Alshikh, Yash Parag Butala","submitted_at":"2024-02-27T14:47:53Z","abstract_excerpt":"For decades, human-computer interaction has fundamentally been manual. Even today, almost all productive work done on the computer necessitates human input at every step. Autonomous virtual agents represent an exciting step in automating many of these menial tasks. Virtual agents would empower users with limited technical proficiency to harness the full possibilities of computer systems. They could also enable the efficient streamlining of numerous computer tasks, ranging from calendar management to complex travel bookings, with minimal human intervention. In this paper, we introduce OmniACT, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17553","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17553/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17553","created_at":"2026-07-05T08:46:22.051479+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17553v3","created_at":"2026-07-05T08:46:22.051479+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17553","created_at":"2026-07-05T08:46:22.051479+00:00"},{"alias_kind":"pith_short_12","alias_value":"SLYYAOPB7T4E","created_at":"2026-07-05T08:46:22.051479+00:00"},{"alias_kind":"pith_short_16","alias_value":"SLYYAOPB7T4EGGMZ","created_at":"2026-07-05T08:46:22.051479+00:00"},{"alias_kind":"pith_short_8","alias_value":"SLYYAOPB","created_at":"2026-07-05T08:46:22.051479+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31270","citing_title":"Learning from Failure: Inference-Time Self-Improvement for Computer-Use Agents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25160","citing_title":"ScaleWoB: Guiding GUI Agents with Coding Agents via Large-Scale Environmental Synthesis","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2406.12373","citing_title":"WebCanvas: Benchmarking Web Agents in Online Environments","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04454","citing_title":"Aguvis: Unified Pure Vision Agents for Autonomous GUI Interaction","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05459","citing_title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14573","citing_title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2410.23218","citing_title":"OS-ATLAS: A Foundation Action Model for Generalist GUI Agents","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07972","citing_title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21375","citing_title":"VLAA-GUI: Knowing When to Stop, Recover, and Search, A Modular Framework for GUI Automation","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB","json":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB.json","graph_json":"https://pith.science/api/pith-number/SLYYAOPB7T4EGGMZPM4SSXDPKB/graph.json","events_json":"https://pith.science/api/pith-number/SLYYAOPB7T4EGGMZPM4SSXDPKB/events.json","paper":"https://pith.science/paper/SLYYAOPB"},"agent_actions":{"view_html":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB","download_json":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB.json","view_paper":"https://pith.science/paper/SLYYAOPB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17553&json=true","fetch_graph":"https://pith.science/api/pith-number/SLYYAOPB7T4EGGMZPM4SSXDPKB/graph.json","fetch_events":"https://pith.science/api/pith-number/SLYYAOPB7T4EGGMZPM4SSXDPKB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB/action/storage_attestation","attest_author":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB/action/author_attestation","sign_citation":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB/action/citation_signature","submit_replication":"https://pith.science/pith/SLYYAOPB7T4EGGMZPM4SSXDPKB/action/replication_record"}},"created_at":"2026-07-05T08:46:22.051479+00:00","updated_at":"2026-07-05T08:46:22.051479+00:00"}