{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6VCT5D7WARX4OT4PX3I6GG64RR","short_pith_number":"pith:6VCT5D7W","schema_version":"1.0","canonical_sha256":"f5453e8ff6046fc74f8fbed1e31bdc8c47f1716a5f3c0dcba4f07dae8ff81879","source":{"kind":"arxiv","id":"2405.00823","version":2},"attestation_state":"computed","paper":{"title":"WorkBench: a Benchmark Dataset for Agents in a Realistic Workplace Setting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.CL","authors_text":"Bertie Vidgen, Olly Styles, Patricio Cerda-Mardini, Sam Miller, Tanaya Guha, Victor Sanchez","submitted_at":"2024-05-01T19:07:03Z","abstract_excerpt":"We introduce WorkBench: a benchmark dataset for evaluating agents' ability to execute tasks in a workplace setting. WorkBench contains a sandbox environment with five databases, 26 tools, and 690 tasks. These tasks represent common business activities, such as sending emails and scheduling meetings. The tasks in WorkBench are challenging as they require planning, tool selection, and often multiple actions. If a task has been successfully executed, one (or more) of the database values may change. The correct outcome for each task is unique and unambiguous, which allows for robust, automated eva"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.00823","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-01T19:07:03Z","cross_cats_sorted":["cs.AI","cs.MA"],"title_canon_sha256":"fc5600aec32d7565a760f8311e55f6712cb479366969b2fb3ba8c5d7a453736c","abstract_canon_sha256":"f1210784f75fa632740eb133c47a4642fae787aa2623e4f4d1a3d8b3af46de87"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:51.376179Z","signature_b64":"Vj0EeX9xmni4j76zbHCWJOssetcEVTUKHkJpChkIbhNFHyCOLeZ0RabSgGUMxyuOxqViGNTK/GW/ll8SbkJzBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5453e8ff6046fc74f8fbed1e31bdc8c47f1716a5f3c0dcba4f07dae8ff81879","last_reissued_at":"2026-07-05T08:51:51.375676Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:51.375676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WorkBench: a Benchmark Dataset for Agents in a Realistic Workplace Setting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.MA"],"primary_cat":"cs.CL","authors_text":"Bertie Vidgen, Olly Styles, Patricio Cerda-Mardini, Sam Miller, Tanaya Guha, Victor Sanchez","submitted_at":"2024-05-01T19:07:03Z","abstract_excerpt":"We introduce WorkBench: a benchmark dataset for evaluating agents' ability to execute tasks in a workplace setting. WorkBench contains a sandbox environment with five databases, 26 tools, and 690 tasks. These tasks represent common business activities, such as sending emails and scheduling meetings. The tasks in WorkBench are challenging as they require planning, tool selection, and often multiple actions. If a task has been successfully executed, one (or more) of the database values may change. The correct outcome for each task is unique and unambiguous, which allows for robust, automated eva"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00823","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.00823/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.00823","created_at":"2026-07-05T08:51:51.375734+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.00823v2","created_at":"2026-07-05T08:51:51.375734+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00823","created_at":"2026-07-05T08:51:51.375734+00:00"},{"alias_kind":"pith_short_12","alias_value":"6VCT5D7WARX4","created_at":"2026-07-05T08:51:51.375734+00:00"},{"alias_kind":"pith_short_16","alias_value":"6VCT5D7WARX4OT4P","created_at":"2026-07-05T08:51:51.375734+00:00"},{"alias_kind":"pith_short_8","alias_value":"6VCT5D7W","created_at":"2026-07-05T08:51:51.375734+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07909","citing_title":"MemToolAgent: Leveraging Memory for Tool Using Agents Based on Environment and User Feedback","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27621","citing_title":"Agents that Matter: Optimizing Multi-Agent LLMs via Removal-Based Attribution","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23127","citing_title":"Managing Procedural Memory in LLM Agents: Control, Adaptation, and Evaluation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2508.07407","citing_title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09539","citing_title":"TacoMAS: Test-Time Co-Evolution of Topology and Capability in LLM-based Multi-Agent Systems","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13564","citing_title":"Memory in the Age of AI Agents","ref_index":297,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR","json":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR.json","graph_json":"https://pith.science/api/pith-number/6VCT5D7WARX4OT4PX3I6GG64RR/graph.json","events_json":"https://pith.science/api/pith-number/6VCT5D7WARX4OT4PX3I6GG64RR/events.json","paper":"https://pith.science/paper/6VCT5D7W"},"agent_actions":{"view_html":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR","download_json":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR.json","view_paper":"https://pith.science/paper/6VCT5D7W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.00823&json=true","fetch_graph":"https://pith.science/api/pith-number/6VCT5D7WARX4OT4PX3I6GG64RR/graph.json","fetch_events":"https://pith.science/api/pith-number/6VCT5D7WARX4OT4PX3I6GG64RR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR/action/storage_attestation","attest_author":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR/action/author_attestation","sign_citation":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR/action/citation_signature","submit_replication":"https://pith.science/pith/6VCT5D7WARX4OT4PX3I6GG64RR/action/replication_record"}},"created_at":"2026-07-05T08:51:51.375734+00:00","updated_at":"2026-07-05T08:51:51.375734+00:00"}