{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4U5ZKBBE235HGM5AUQV3A2NM3B","short_pith_number":"pith:4U5ZKBBE","schema_version":"1.0","canonical_sha256":"e53b950424d6fa7333a0a42bb069acd8712760cb3bb8429a2e424e37fb284ac9","source":{"kind":"arxiv","id":"2503.04957","version":1},"attestation_state":"computed","paper":{"title":"SafeArena: Evaluating the Safety of Autonomous Web Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ada Defne Tur, Alejandra Zambrano, Arkil Patel, Esin Durmus, Karolina Sta\\'nczak, Nicholas Meade, Siva Reddy, Spandana Gella, Xing Han L\\`u","submitted_at":"2025-03-06T20:43:14Z","abstract_excerpt":"LLM-based agents are becoming increasingly proficient at solving web-based tasks. With this capability comes a greater risk of misuse for malicious purposes, such as posting misinformation in an online forum or selling illicit substances on a website. To evaluate these risks, we propose SafeArena, the first benchmark to focus on the deliberate misuse of web agents. SafeArena comprises 250 safe and 250 harmful tasks across four websites. We classify the harmful tasks into five harm categories -- misinformation, illegal activity, harassment, cybercrime, and social bias, designed to assess realis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.04957","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-06T20:43:14Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"936858ab191e8e88b8653e757813a3e8e23af10ac645a628c8a3395a242d0894","abstract_canon_sha256":"503dd286dde0b38709d991ebef4e7718d907eed294f54918d76aa14153d8de4c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:17.320746Z","signature_b64":"Rvnob3ZvUlZdtlkH7Llj/5VJofRUX3GgRTMorf+Y2W5Z62yVQYrtUxk69bxg+xzRKnR0aRVRkhQOQe0gE+v4Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e53b950424d6fa7333a0a42bb069acd8712760cb3bb8429a2e424e37fb284ac9","last_reissued_at":"2026-07-05T10:26:17.320259Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:17.320259Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafeArena: Evaluating the Safety of Autonomous Web Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ada Defne Tur, Alejandra Zambrano, Arkil Patel, Esin Durmus, Karolina Sta\\'nczak, Nicholas Meade, Siva Reddy, Spandana Gella, Xing Han L\\`u","submitted_at":"2025-03-06T20:43:14Z","abstract_excerpt":"LLM-based agents are becoming increasingly proficient at solving web-based tasks. With this capability comes a greater risk of misuse for malicious purposes, such as posting misinformation in an online forum or selling illicit substances on a website. To evaluate these risks, we propose SafeArena, the first benchmark to focus on the deliberate misuse of web agents. SafeArena comprises 250 safe and 250 harmful tasks across four websites. We classify the harmful tasks into five harm categories -- misinformation, illegal activity, harassment, cybercrime, and social bias, designed to assess realis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.04957","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.04957/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.04957","created_at":"2026-07-05T10:26:17.320318+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.04957v1","created_at":"2026-07-05T10:26:17.320318+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.04957","created_at":"2026-07-05T10:26:17.320318+00:00"},{"alias_kind":"pith_short_12","alias_value":"4U5ZKBBE235H","created_at":"2026-07-05T10:26:17.320318+00:00"},{"alias_kind":"pith_short_16","alias_value":"4U5ZKBBE235HGM5A","created_at":"2026-07-05T10:26:17.320318+00:00"},{"alias_kind":"pith_short_8","alias_value":"4U5ZKBBE","created_at":"2026-07-05T10:26:17.320318+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23049","citing_title":"PhoneBuddy: Training Open Models for Agentic Phone Use","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19409","citing_title":"OpenRath: Session-Centered Runtime State for Agent Systems","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05647","citing_title":"Coding with \"Enemy\": Can Human Developers Detect AI Agent Sabotage?","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00341","citing_title":"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30755","citing_title":"Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23772","citing_title":"PageGuide: Browser extension to assist users in navigating a webpage and locating information","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30693","citing_title":"Triaging Threats to Specialized Guardrails","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10073","citing_title":"SecureWebArena: A Holistic Security Evaluation Benchmark for LVLM-based Web Agents","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":247,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11030","citing_title":"An Executable Benchmarking Suite for Tool-Using Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23772","citing_title":"PageGuide: Browser extension to assist users in navigating a webpage and locating information","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06367","citing_title":"WebSP-Eval: Evaluating Web Agents on Website Security and Privacy Tasks","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B","json":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B.json","graph_json":"https://pith.science/api/pith-number/4U5ZKBBE235HGM5AUQV3A2NM3B/graph.json","events_json":"https://pith.science/api/pith-number/4U5ZKBBE235HGM5AUQV3A2NM3B/events.json","paper":"https://pith.science/paper/4U5ZKBBE"},"agent_actions":{"view_html":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B","download_json":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B.json","view_paper":"https://pith.science/paper/4U5ZKBBE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.04957&json=true","fetch_graph":"https://pith.science/api/pith-number/4U5ZKBBE235HGM5AUQV3A2NM3B/graph.json","fetch_events":"https://pith.science/api/pith-number/4U5ZKBBE235HGM5AUQV3A2NM3B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B/action/storage_attestation","attest_author":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B/action/author_attestation","sign_citation":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B/action/citation_signature","submit_replication":"https://pith.science/pith/4U5ZKBBE235HGM5AUQV3A2NM3B/action/replication_record"}},"created_at":"2026-07-05T10:26:17.320318+00:00","updated_at":"2026-07-05T10:26:17.320318+00:00"}