{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RFZTG4EWXJSEHH4VCKNEVWOLRE","short_pith_number":"pith:RFZTG4EW","schema_version":"1.0","canonical_sha256":"8973337096ba64439f95129a4ad9cb893422fa74d48f1b70928aece7f0dbd731","source":{"kind":"arxiv","id":"2502.18356","version":1},"attestation_state":"computed","paper":{"title":"WebGames: Challenging General-Purpose Web-Browsing AI Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Andy Toulis, Filippos Christianos, Fraser Greenlee, George Thomas, Jikun Kang, Marvin Purtorab, Wenqi Wu","submitted_at":"2025-02-25T16:45:08Z","abstract_excerpt":"We introduce WebGames, a comprehensive benchmark suite designed to evaluate general-purpose web-browsing AI agents through a collection of 50+ interactive challenges. These challenges are specifically crafted to be straightforward for humans while systematically testing the limitations of current AI systems across fundamental browser interactions, advanced input processing, cognitive tasks, workflow automation, and interactive entertainment. Our framework eliminates external dependencies through a hermetic testing environment, ensuring reproducible evaluation with verifiable ground-truth solut"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18356","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-25T16:45:08Z","cross_cats_sorted":[],"title_canon_sha256":"f721c34984276127962f9b6d8b113ec033a8a4ba8ba9ab55a17ed7dad4dc487b","abstract_canon_sha256":"191f20e35d5d7b039098a50225e481c9bae488ccc91d290a7e3686da1d2d38ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:52.660253Z","signature_b64":"EjJHwrND759g/JTJTntobF/M3j4VuPan69RkX2SvcW/hfMLuZTJAHYAI7mtbjS2A9+LOvG1YbrpXGaOv73BtDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8973337096ba64439f95129a4ad9cb893422fa74d48f1b70928aece7f0dbd731","last_reissued_at":"2026-07-05T10:19:52.659775Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:52.659775Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WebGames: Challenging General-Purpose Web-Browsing AI Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Andy Toulis, Filippos Christianos, Fraser Greenlee, George Thomas, Jikun Kang, Marvin Purtorab, Wenqi Wu","submitted_at":"2025-02-25T16:45:08Z","abstract_excerpt":"We introduce WebGames, a comprehensive benchmark suite designed to evaluate general-purpose web-browsing AI agents through a collection of 50+ interactive challenges. These challenges are specifically crafted to be straightforward for humans while systematically testing the limitations of current AI systems across fundamental browser interactions, advanced input processing, cognitive tasks, workflow automation, and interactive entertainment. Our framework eliminates external dependencies through a hermetic testing environment, ensuring reproducible evaluation with verifiable ground-truth solut"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18356","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18356","created_at":"2026-07-05T10:19:52.659835+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18356v1","created_at":"2026-07-05T10:19:52.659835+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18356","created_at":"2026-07-05T10:19:52.659835+00:00"},{"alias_kind":"pith_short_12","alias_value":"RFZTG4EWXJSE","created_at":"2026-07-05T10:19:52.659835+00:00"},{"alias_kind":"pith_short_16","alias_value":"RFZTG4EWXJSEHH4V","created_at":"2026-07-05T10:19:52.659835+00:00"},{"alias_kind":"pith_short_8","alias_value":"RFZTG4EW","created_at":"2026-07-05T10:19:52.659835+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02449","citing_title":"HLL: Can Agents Cross Humanity's Last Line of Verification?","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE","json":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE.json","graph_json":"https://pith.science/api/pith-number/RFZTG4EWXJSEHH4VCKNEVWOLRE/graph.json","events_json":"https://pith.science/api/pith-number/RFZTG4EWXJSEHH4VCKNEVWOLRE/events.json","paper":"https://pith.science/paper/RFZTG4EW"},"agent_actions":{"view_html":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE","download_json":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE.json","view_paper":"https://pith.science/paper/RFZTG4EW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18356&json=true","fetch_graph":"https://pith.science/api/pith-number/RFZTG4EWXJSEHH4VCKNEVWOLRE/graph.json","fetch_events":"https://pith.science/api/pith-number/RFZTG4EWXJSEHH4VCKNEVWOLRE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE/action/storage_attestation","attest_author":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE/action/author_attestation","sign_citation":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE/action/citation_signature","submit_replication":"https://pith.science/pith/RFZTG4EWXJSEHH4VCKNEVWOLRE/action/replication_record"}},"created_at":"2026-07-05T10:19:52.659835+00:00","updated_at":"2026-07-05T10:19:52.659835+00:00"}