{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3G2LZFUEH3VN63LJS4KQVMA4IW","short_pith_number":"pith:3G2LZFUE","schema_version":"1.0","canonical_sha256":"d9b4bc96843eeadf6d6997150ab01c45bbb9c75a99179d398755d136346f20d0","source":{"kind":"arxiv","id":"2407.19056","version":1},"attestation_state":"computed","paper":{"title":"OfficeBench: Benchmarking Language Agents across Multiple Applications for Office Automation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Yuchen Lin, Da Yin, Jingbo Shang, Li Zhong, Yuedong Cui, Zilong Wang, Zimin Zhang","submitted_at":"2024-07-26T19:27:17Z","abstract_excerpt":"Office automation significantly enhances human productivity by automatically finishing routine tasks in the workflow. Beyond the basic information extraction studied in much of the prior document AI literature, the office automation research should be extended to more realistic office tasks which require to integrate various information sources in the office system and produce outputs through a series of decision-making processes. We introduce OfficeBench, one of the first office automation benchmarks for evaluating current LLM agents' capability to address office tasks in realistic office wor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19056","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-26T19:27:17Z","cross_cats_sorted":[],"title_canon_sha256":"3e94fa5853ef6d9f456eaa1b44cef57bdf5d573c892cb524b05a403f6690b03b","abstract_canon_sha256":"99b95e7e63e76acf8ba4992582a80b6fbaaba65f11b72b43bd2fa92995535e1d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:13.203017Z","signature_b64":"0rETKUrMOO3wJMulyjIIvxhwA1Ol516a4ruGdaNOo7eEyRytKYWHKA57CozlMWyFVFNvr/h24mBIBASMOgXUAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d9b4bc96843eeadf6d6997150ab01c45bbb9c75a99179d398755d136346f20d0","last_reissued_at":"2026-07-05T08:49:13.202533Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:13.202533Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OfficeBench: Benchmarking Language Agents across Multiple Applications for Office Automation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Yuchen Lin, Da Yin, Jingbo Shang, Li Zhong, Yuedong Cui, Zilong Wang, Zimin Zhang","submitted_at":"2024-07-26T19:27:17Z","abstract_excerpt":"Office automation significantly enhances human productivity by automatically finishing routine tasks in the workflow. Beyond the basic information extraction studied in much of the prior document AI literature, the office automation research should be extended to more realistic office tasks which require to integrate various information sources in the office system and produce outputs through a series of decision-making processes. We introduce OfficeBench, one of the first office automation benchmarks for evaluating current LLM agents' capability to address office tasks in realistic office wor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19056","created_at":"2026-07-05T08:49:13.202592+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19056v1","created_at":"2026-07-05T08:49:13.202592+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19056","created_at":"2026-07-05T08:49:13.202592+00:00"},{"alias_kind":"pith_short_12","alias_value":"3G2LZFUEH3VN","created_at":"2026-07-05T08:49:13.202592+00:00"},{"alias_kind":"pith_short_16","alias_value":"3G2LZFUEH3VN63LJ","created_at":"2026-07-05T08:49:13.202592+00:00"},{"alias_kind":"pith_short_8","alias_value":"3G2LZFUE","created_at":"2026-07-05T08:49:13.202592+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.06008","citing_title":"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06008","citing_title":"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30907","citing_title":"BlueFin: Benchmarking LLM Agents on Financial Spreadsheets","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28480","citing_title":"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29537","citing_title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06708","citing_title":"Signal-Driven Observation for Long-Horizon Web Agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2501.16150","citing_title":"A Comprehensive Survey of Agents for Computer Use: Foundations, Challenges, and Future Directions","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13168","citing_title":"Finch: Benchmarking Finance & Accounting across Spreadsheet-Centric Enterprise Workflows","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13880","citing_title":"PREPING: Building Agent Memory without Tasks","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10384","citing_title":"Agentic Performance at the Edge: Insights from Benchmarking","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05172","citing_title":"ClawsBench: Evaluating Capability and Safety of LLM Productivity Agents in Simulated Workspaces","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02729","citing_title":"Augmenting Interface Usability Heuristics for Reliable Computer-Use Agents","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW","json":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW.json","graph_json":"https://pith.science/api/pith-number/3G2LZFUEH3VN63LJS4KQVMA4IW/graph.json","events_json":"https://pith.science/api/pith-number/3G2LZFUEH3VN63LJS4KQVMA4IW/events.json","paper":"https://pith.science/paper/3G2LZFUE"},"agent_actions":{"view_html":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW","download_json":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW.json","view_paper":"https://pith.science/paper/3G2LZFUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19056&json=true","fetch_graph":"https://pith.science/api/pith-number/3G2LZFUEH3VN63LJS4KQVMA4IW/graph.json","fetch_events":"https://pith.science/api/pith-number/3G2LZFUEH3VN63LJS4KQVMA4IW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW/action/storage_attestation","attest_author":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW/action/author_attestation","sign_citation":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW/action/citation_signature","submit_replication":"https://pith.science/pith/3G2LZFUEH3VN63LJS4KQVMA4IW/action/replication_record"}},"created_at":"2026-07-05T08:49:13.202592+00:00","updated_at":"2026-07-05T08:49:13.202592+00:00"}