{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3Z2ODSFKUS5OKWXYAWKFR4NLPQ","short_pith_number":"pith:3Z2ODSFK","schema_version":"1.0","canonical_sha256":"de74e1c8aaa4bae55af8059458f1ab7c15c2d62ee28312c83b107d5610865656","source":{"kind":"arxiv","id":"2502.05352","version":1},"attestation_state":"computed","paper":{"title":"ITBench: Evaluating AI Agents across Diverse Real-World IT Automation Tasks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.DC","cs.MA"],"primary_cat":"cs.AI","authors_text":"(2) University of Illinois at Urbana-Champaign), Ameet Rahane (1), Amit Paradkar (1), Anca Sailer (1), Bekir O. Turkkan (1), Bhavya Bhavya (1), Carlos Fonseca (1), Chandrasekhar Narayanaswami (1), Daby Sow (1), Debanjana Kar (1), Divya Pathak (1), Felix George (1), Gerard Vanloo (1), Harshit Kumar (1), Hirokuni Kitahara (1), Jackson Clark (2), Jae-wook Ahn (1), Laura Shwartz (1), Lav R. Varshney (2), Michael Nidd (1), Mudit Verma (1), Naoki Abe (1), Nicholas C. M. Fuller (1), Noah Zheutlin (1), Oishik Chatterjee (1), Pavankumar Murali (1), Pooja Aggarwal (1), Pranjal Gupta (1), Prateeti Mohapatra (1), Pratibha Moogi (1), Rohan Arora (1), Rong Lee (1), Ruchi Mahindru (1), Ruchir Puri (1) ((1) IBM, Saki Takano (1), Saurabh Jha (1), Suranjana Samanta (1), Takumi Yanagawa (1), Tianyin Xu (2), Ting Dai (1), Xinbo Wu (2), Yinfang Chen (2), Yu Deng (1), Yuji Watanabe (1)","submitted_at":"2025-02-07T21:46:52Z","abstract_excerpt":"Realizing the vision of using AI agents to automate critical IT tasks depends on the ability to measure and understand effectiveness of proposed solutions. We introduce ITBench, a framework that offers a systematic methodology for benchmarking AI agents to address real-world IT automation tasks. Our initial release targets three key areas: Site Reliability Engineering (SRE), Compliance and Security Operations (CISO), and Financial Operations (FinOps). The design enables AI researchers to understand the challenges and opportunities of AI agents for IT automation with push-button workflows and i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05352","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-07T21:46:52Z","cross_cats_sorted":["cs.DC","cs.MA"],"title_canon_sha256":"5caea945e13939fd21b16cb4729f860535588d760c7c42a460c673e08e386617","abstract_canon_sha256":"0dd3d1930f53b9d65310e33d33d1207edae8ba873725d3ca13598e2bb2f33d53"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:36.763254Z","signature_b64":"vQbxji9qyXQPbhrc8WMQtsIbmhwa24gE09z+Bou8qJUY3MSi/jMovr8HUJ7/WnL+mMgbqbpsI1oTcfpfvh7NBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de74e1c8aaa4bae55af8059458f1ab7c15c2d62ee28312c83b107d5610865656","last_reissued_at":"2026-07-05T10:11:36.762617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:36.762617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ITBench: Evaluating AI Agents across Diverse Real-World IT Automation Tasks","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.DC","cs.MA"],"primary_cat":"cs.AI","authors_text":"(2) University of Illinois at Urbana-Champaign), Ameet Rahane (1), Amit Paradkar (1), Anca Sailer (1), Bekir O. Turkkan (1), Bhavya Bhavya (1), Carlos Fonseca (1), Chandrasekhar Narayanaswami (1), Daby Sow (1), Debanjana Kar (1), Divya Pathak (1), Felix George (1), Gerard Vanloo (1), Harshit Kumar (1), Hirokuni Kitahara (1), Jackson Clark (2), Jae-wook Ahn (1), Laura Shwartz (1), Lav R. Varshney (2), Michael Nidd (1), Mudit Verma (1), Naoki Abe (1), Nicholas C. M. Fuller (1), Noah Zheutlin (1), Oishik Chatterjee (1), Pavankumar Murali (1), Pooja Aggarwal (1), Pranjal Gupta (1), Prateeti Mohapatra (1), Pratibha Moogi (1), Rohan Arora (1), Rong Lee (1), Ruchi Mahindru (1), Ruchir Puri (1) ((1) IBM, Saki Takano (1), Saurabh Jha (1), Suranjana Samanta (1), Takumi Yanagawa (1), Tianyin Xu (2), Ting Dai (1), Xinbo Wu (2), Yinfang Chen (2), Yu Deng (1), Yuji Watanabe (1)","submitted_at":"2025-02-07T21:46:52Z","abstract_excerpt":"Realizing the vision of using AI agents to automate critical IT tasks depends on the ability to measure and understand effectiveness of proposed solutions. We introduce ITBench, a framework that offers a systematic methodology for benchmarking AI agents to address real-world IT automation tasks. Our initial release targets three key areas: Site Reliability Engineering (SRE), Compliance and Security Operations (CISO), and Financial Operations (FinOps). The design enables AI researchers to understand the challenges and opportunities of AI agents for IT automation with push-button workflows and i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05352","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05352","created_at":"2026-07-05T10:11:36.762683+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05352v1","created_at":"2026-07-05T10:11:36.762683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05352","created_at":"2026-07-05T10:11:36.762683+00:00"},{"alias_kind":"pith_short_12","alias_value":"3Z2ODSFKUS5O","created_at":"2026-07-05T10:11:36.762683+00:00"},{"alias_kind":"pith_short_16","alias_value":"3Z2ODSFKUS5OKWXY","created_at":"2026-07-05T10:11:36.762683+00:00"},{"alias_kind":"pith_short_8","alias_value":"3Z2ODSFK","created_at":"2026-07-05T10:11:36.762683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08590","citing_title":"Auditable Graph-Guided Root Cause Analysis for Kubernetes Incidents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29159","citing_title":"Pooled Leaderboards Hide System-Specific Winners: A Reporting-Protocol Audit of Offline Root-Cause Analysis Benchmarks","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04375","citing_title":"Experiment-as-Code Labs: A Declarative Stack for AI-Driven Scientific Discovery","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15425","citing_title":"Runtime-Structured Task Decomposition for Agentic Coding Systems","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07062","citing_title":"From Assistance to Agency: Rethinking Autonomy and Control in CI/CD Pipelines","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15597","citing_title":"LLMs Corrupt Your Documents When You Delegate","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04375","citing_title":"Experiment-as-Code Labs: A Declarative Stack for AI-Driven Scientific Discovery","ref_index":136,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ","json":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ.json","graph_json":"https://pith.science/api/pith-number/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/graph.json","events_json":"https://pith.science/api/pith-number/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/events.json","paper":"https://pith.science/paper/3Z2ODSFK"},"agent_actions":{"view_html":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ","download_json":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ.json","view_paper":"https://pith.science/paper/3Z2ODSFK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05352&json=true","fetch_graph":"https://pith.science/api/pith-number/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/graph.json","fetch_events":"https://pith.science/api/pith-number/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/action/storage_attestation","attest_author":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/action/author_attestation","sign_citation":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/action/citation_signature","submit_replication":"https://pith.science/pith/3Z2ODSFKUS5OKWXYAWKFR4NLPQ/action/replication_record"}},"created_at":"2026-07-05T10:11:36.762683+00:00","updated_at":"2026-07-05T10:11:36.762683+00:00"}