{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4DDBIT3ZDIUUTFXVEHSCBXDNC2","short_pith_number":"pith:4DDBIT3Z","schema_version":"1.0","canonical_sha256":"e0c6144f791a294996f521e420dc6d16b4834fd37f7aa50cc5112a4319674f83","source":{"kind":"arxiv","id":"2412.05467","version":4},"attestation_state":"computed","paper":{"title":"The BrowserGym Ecosystem for Web Agent Research","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.LG","authors_text":"Alexandre Drouin, Alexandre Lacoste, Dehan Kong, Frank F. Xu, Graham Neubig, Lawrence Keunho Jang, L\\'eo Boisvert, Massimo Caccia, Maxime Gasse, Megh Thakkar, Nicolas Chapados, Ori Yoran, Quentin Cappart, Rim Assouel, Ruslan Salakhutdinov, Sahar Omidi Shayegan, Siva Reddy, Thibault Le Sellier De Chezelles, Tom Marty, Xing Han L\\`u","submitted_at":"2024-12-06T23:43:59Z","abstract_excerpt":"The BrowserGym ecosystem addresses the growing need for efficient evaluation and benchmarking of web agents, particularly those leveraging automation and Large Language Models (LLMs). Many existing benchmarks suffer from fragmentation and inconsistent evaluation methodologies, making it challenging to achieve reliable comparisons and reproducible results. In an earlier work, Drouin et al. (2024) introduced BrowserGym which aims to solve this by providing a unified, gym-like environment with well-defined observation and action spaces, facilitating standardized evaluation across diverse benchmar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.05467","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-06T23:43:59Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"94f5f67b767d63c054d97f342b7ff6af5054f78d866c1d83f68fd539c3a0899e","abstract_canon_sha256":"e84c9afaa46629651a8313b64259ed794a93c5d58f23dc7272e74f36e81cd20e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:27.009620Z","signature_b64":"4YOwizqttDTkmWxS/GopENduRUQJsr4BZJkfPdgVhFsBZJ4iiYB7/chjk+TtCNWC0ZLKd+LRVSIpvsfzzyeTDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e0c6144f791a294996f521e420dc6d16b4834fd37f7aa50cc5112a4319674f83","last_reissued_at":"2026-07-05T10:21:27.009069Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:27.009069Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The BrowserGym Ecosystem for Web Agent Research","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.LG","authors_text":"Alexandre Drouin, Alexandre Lacoste, Dehan Kong, Frank F. Xu, Graham Neubig, Lawrence Keunho Jang, L\\'eo Boisvert, Massimo Caccia, Maxime Gasse, Megh Thakkar, Nicolas Chapados, Ori Yoran, Quentin Cappart, Rim Assouel, Ruslan Salakhutdinov, Sahar Omidi Shayegan, Siva Reddy, Thibault Le Sellier De Chezelles, Tom Marty, Xing Han L\\`u","submitted_at":"2024-12-06T23:43:59Z","abstract_excerpt":"The BrowserGym ecosystem addresses the growing need for efficient evaluation and benchmarking of web agents, particularly those leveraging automation and Large Language Models (LLMs). Many existing benchmarks suffer from fragmentation and inconsistent evaluation methodologies, making it challenging to achieve reliable comparisons and reproducible results. In an earlier work, Drouin et al. (2024) introduced BrowserGym which aims to solve this by providing a unified, gym-like environment with well-defined observation and action spaces, facilitating standardized evaluation across diverse benchmar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.05467","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.05467/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.05467","created_at":"2026-07-05T10:21:27.009131+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.05467v4","created_at":"2026-07-05T10:21:27.009131+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.05467","created_at":"2026-07-05T10:21:27.009131+00:00"},{"alias_kind":"pith_short_12","alias_value":"4DDBIT3ZDIUU","created_at":"2026-07-05T10:21:27.009131+00:00"},{"alias_kind":"pith_short_16","alias_value":"4DDBIT3ZDIUUTFXV","created_at":"2026-07-05T10:21:27.009131+00:00"},{"alias_kind":"pith_short_8","alias_value":"4DDBIT3Z","created_at":"2026-07-05T10:21:27.009131+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08768","citing_title":"UniClawBench: A Universal Benchmark for Proactive Agents on Real-World Tasks","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00007","citing_title":"BaRA: Budget-constrained and Reliable Web Data Collection Agent","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20683","citing_title":"From Question Answering to Task Completion: A Survey on Agent System and Harness Design","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14517","citing_title":"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13608","citing_title":"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09447","citing_title":"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07805","citing_title":"Beyond Goodhart's Law: A Dynamic Benchmark for Evaluating Compliance in Multi-Agent Systems","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05342","citing_title":"SentinelBench: A Benchmark for Long-Running Monitoring Agents","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05233","citing_title":"Domain-Conditioned Safety in Frontier Computer-Using Agents: A 793-Episode Browser Benchmark, a Coding-Domain Cross-Reference, and a Reproducibility Audit of Recent Red-Teaming","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13631","citing_title":"ProjGuard: Safety Monitoring for Computer-Use Agents via Low-Dimensional Projections","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29537","citing_title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28787","citing_title":"Do Data Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06708","citing_title":"Signal-Driven Observation for Long-Horizon Web Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17637","citing_title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17637","citing_title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19219","citing_title":"SimGym: A Framework for A/B Test Simulation in E-Commerce with Traffic-Grounded VLM Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08136","citing_title":"EconWebArena: Benchmarking Autonomous Agents on Economic Tasks in Realistic Web Environments","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13024","citing_title":"WebMall -- A Multi-Shop Benchmark for Evaluating Web Agents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15832","citing_title":"A Functionality-Grounded Benchmark for Evaluating Web Agents in E-commerce Domains","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09505","citing_title":"Combating the Memory Walls: Optimization Pathways for Long-Context Agentic LLM Inference","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2511.23281","citing_title":"MCP vs RAG vs NLWeb vs HTML: A Comparison of the Effectiveness and Efficiency of Different Agent Interfaces to the Web (Technical Report)","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12634","citing_title":"MobiBench: Multi-Branch, Modular Benchmark for Mobile GUI Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05044","citing_title":"WebFactory: Automated Compression of Foundational Language Intelligence into Grounded Web Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03515","citing_title":"Inside the Scaffold: A Source-Code Taxonomy of Coding Agent Architectures","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2","json":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2.json","graph_json":"https://pith.science/api/pith-number/4DDBIT3ZDIUUTFXVEHSCBXDNC2/graph.json","events_json":"https://pith.science/api/pith-number/4DDBIT3ZDIUUTFXVEHSCBXDNC2/events.json","paper":"https://pith.science/paper/4DDBIT3Z"},"agent_actions":{"view_html":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2","download_json":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2.json","view_paper":"https://pith.science/paper/4DDBIT3Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.05467&json=true","fetch_graph":"https://pith.science/api/pith-number/4DDBIT3ZDIUUTFXVEHSCBXDNC2/graph.json","fetch_events":"https://pith.science/api/pith-number/4DDBIT3ZDIUUTFXVEHSCBXDNC2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2/action/storage_attestation","attest_author":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2/action/author_attestation","sign_citation":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2/action/citation_signature","submit_replication":"https://pith.science/pith/4DDBIT3ZDIUUTFXVEHSCBXDNC2/action/replication_record"}},"created_at":"2026-07-05T10:21:27.009131+00:00","updated_at":"2026-07-05T10:21:27.009131+00:00"}