{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7XZPVRFY4DOZP3ZXS67X63ET67","short_pith_number":"pith:7XZPVRFY","schema_version":"1.0","canonical_sha256":"fdf2fac4b8e0dd97ef3797bf7f6c93f7c8e44ddf72d7ed0c98f43244d02a722c","source":{"kind":"arxiv","id":"2406.01623","version":1},"attestation_state":"computed","paper":{"title":"WebSuite: Systematically Evaluating Why Web Agents Fail","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Eric Li, Jim Waldo","submitted_at":"2024-06-01T00:32:26Z","abstract_excerpt":"We describe WebSuite, the first diagnostic benchmark for generalist web agents, designed to systematically evaluate why agents fail. Advances in AI have led to the rise of numerous web agents that autonomously operate a browser to complete tasks. However, most existing benchmarks focus on strictly measuring whether an agent can or cannot complete a task, without giving insight on why. In this paper, we 1) develop a taxonomy of web actions to facilitate identifying common failure patterns, and 2) create an extensible benchmark suite to assess agents' performance on our taxonomized actions. This"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.01623","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-06-01T00:32:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3907f792ca37f984c499d2f6ebe766135a7a5256696adfa7c39885f062f86208","abstract_canon_sha256":"b5c1aaa514fbb3bf1e8282927d34128b1210158a4e0daaa1afecee03b9d6124d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:45.622940Z","signature_b64":"G0WEk3BlCVBSujhNdqtxACijJvoEeZ7m7qkZUacQWGQphmTFsRoRVmhB5mtD1bcQNG4yp/RFYXylTu4ogKSDCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdf2fac4b8e0dd97ef3797bf7f6c93f7c8e44ddf72d7ed0c98f43244d02a722c","last_reissued_at":"2026-07-05T08:26:45.622391Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:45.622391Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WebSuite: Systematically Evaluating Why Web Agents Fail","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Eric Li, Jim Waldo","submitted_at":"2024-06-01T00:32:26Z","abstract_excerpt":"We describe WebSuite, the first diagnostic benchmark for generalist web agents, designed to systematically evaluate why agents fail. Advances in AI have led to the rise of numerous web agents that autonomously operate a browser to complete tasks. However, most existing benchmarks focus on strictly measuring whether an agent can or cannot complete a task, without giving insight on why. In this paper, we 1) develop a taxonomy of web actions to facilitate identifying common failure patterns, and 2) create an extensible benchmark suite to assess agents' performance on our taxonomized actions. This"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.01623","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.01623/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.01623","created_at":"2026-07-05T08:26:45.622478+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.01623v1","created_at":"2026-07-05T08:26:45.622478+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.01623","created_at":"2026-07-05T08:26:45.622478+00:00"},{"alias_kind":"pith_short_12","alias_value":"7XZPVRFY4DOZ","created_at":"2026-07-05T08:26:45.622478+00:00"},{"alias_kind":"pith_short_16","alias_value":"7XZPVRFY4DOZP3ZX","created_at":"2026-07-05T08:26:45.622478+00:00"},{"alias_kind":"pith_short_8","alias_value":"7XZPVRFY","created_at":"2026-07-05T08:26:45.622478+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25760","citing_title":"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04468","citing_title":"Magentic-One: A Generalist Multi-Agent System for Solving Complex Tasks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15808","citing_title":"Inference-Time Scaling of Verification: Self-Evolving Deep Research Agents via Test-Time Rubric-Guided Verification","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15808","citing_title":"Inference-Time Scaling of Verification: Self-Evolving Deep Research Agents via Test-Time Rubric-Guided Verification","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67","json":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67.json","graph_json":"https://pith.science/api/pith-number/7XZPVRFY4DOZP3ZXS67X63ET67/graph.json","events_json":"https://pith.science/api/pith-number/7XZPVRFY4DOZP3ZXS67X63ET67/events.json","paper":"https://pith.science/paper/7XZPVRFY"},"agent_actions":{"view_html":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67","download_json":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67.json","view_paper":"https://pith.science/paper/7XZPVRFY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.01623&json=true","fetch_graph":"https://pith.science/api/pith-number/7XZPVRFY4DOZP3ZXS67X63ET67/graph.json","fetch_events":"https://pith.science/api/pith-number/7XZPVRFY4DOZP3ZXS67X63ET67/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67/action/storage_attestation","attest_author":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67/action/author_attestation","sign_citation":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67/action/citation_signature","submit_replication":"https://pith.science/pith/7XZPVRFY4DOZP3ZXS67X63ET67/action/replication_record"}},"created_at":"2026-07-05T08:26:45.622478+00:00","updated_at":"2026-07-05T08:26:45.622478+00:00"}