{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VXDOLB53C5Q4KHDEUJJK7CC55K","short_pith_number":"pith:VXDOLB53","schema_version":"1.0","canonical_sha256":"adc6e587bb1761c51c64a252af885deaa216898be6184d10f30efc882aa74c6e","source":{"kind":"arxiv","id":"2411.02305","version":2},"attestation_state":"computed","paper":{"title":"CRMArena: Understanding the Capacity of LLM Agents to Perform Professional CRM Tasks in Realistic Environments","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akshara Prabhakar, Caiming Xiong, Chien-Sheng Wu, Huan Wang, Kung-Hsiang Huang, Philippe Laban, Sidharth Dhawan, Silvio Savarese, Yixin Mao","submitted_at":"2024-11-04T17:30:51Z","abstract_excerpt":"Customer Relationship Management (CRM) systems are vital for modern enterprises, providing a foundation for managing customer interactions and data. Integrating AI agents into CRM systems can automate routine processes and enhance personalized service. However, deploying and evaluating these agents is challenging due to the lack of realistic benchmarks that reflect the complexity of real-world CRM tasks. To address this issue, we introduce CRMArena, a novel benchmark designed to evaluate AI agents on realistic tasks grounded in professional work environments. Following guidance from CRM expert"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02305","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-04T17:30:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c17d9fd2a24c233501cf5cb31d0d9f219a10849518f3c52610c1028dd4a70035","abstract_canon_sha256":"a7f2a7e2c0f6918162c4be3b9b15c1ed53e0026d1065a1e3936cf1bb0c8be2c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:47.657993Z","signature_b64":"zA0us4wtahSo0Q6tmJX1lTk0/2ub9Uf7QoTuc7jbhGUe9NsFYaOY0b3wWcvxi6WuWrbrYYcSxl5kdwwjbci9Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"adc6e587bb1761c51c64a252af885deaa216898be6184d10f30efc882aa74c6e","last_reissued_at":"2026-07-05T10:14:47.657453Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:47.657453Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CRMArena: Understanding the Capacity of LLM Agents to Perform Professional CRM Tasks in Realistic Environments","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akshara Prabhakar, Caiming Xiong, Chien-Sheng Wu, Huan Wang, Kung-Hsiang Huang, Philippe Laban, Sidharth Dhawan, Silvio Savarese, Yixin Mao","submitted_at":"2024-11-04T17:30:51Z","abstract_excerpt":"Customer Relationship Management (CRM) systems are vital for modern enterprises, providing a foundation for managing customer interactions and data. Integrating AI agents into CRM systems can automate routine processes and enhance personalized service. However, deploying and evaluating these agents is challenging due to the lack of realistic benchmarks that reflect the complexity of real-world CRM tasks. To address this issue, we introduce CRMArena, a novel benchmark designed to evaluate AI agents on realistic tasks grounded in professional work environments. Following guidance from CRM expert"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02305","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02305/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02305","created_at":"2026-07-05T10:14:47.657526+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02305v2","created_at":"2026-07-05T10:14:47.657526+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02305","created_at":"2026-07-05T10:14:47.657526+00:00"},{"alias_kind":"pith_short_12","alias_value":"VXDOLB53C5Q4","created_at":"2026-07-05T10:14:47.657526+00:00"},{"alias_kind":"pith_short_16","alias_value":"VXDOLB53C5Q4KHDE","created_at":"2026-07-05T10:14:47.657526+00:00"},{"alias_kind":"pith_short_8","alias_value":"VXDOLB53","created_at":"2026-07-05T10:14:47.657526+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05391","citing_title":"Human oversight of agentic systems in practice: Examining the oversight work, challenges, and heuristics of developers using software agents","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08136","citing_title":"EconWebArena: Benchmarking Autonomous Agents on Economic Tasks in Realistic Web Environments","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08761","citing_title":"Beyond the All-in-One Agent: Benchmarking Role-Specialized Multi-Agent Collaboration in Enterprise Workflows","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K","json":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K.json","graph_json":"https://pith.science/api/pith-number/VXDOLB53C5Q4KHDEUJJK7CC55K/graph.json","events_json":"https://pith.science/api/pith-number/VXDOLB53C5Q4KHDEUJJK7CC55K/events.json","paper":"https://pith.science/paper/VXDOLB53"},"agent_actions":{"view_html":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K","download_json":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K.json","view_paper":"https://pith.science/paper/VXDOLB53","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02305&json=true","fetch_graph":"https://pith.science/api/pith-number/VXDOLB53C5Q4KHDEUJJK7CC55K/graph.json","fetch_events":"https://pith.science/api/pith-number/VXDOLB53C5Q4KHDEUJJK7CC55K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K/action/storage_attestation","attest_author":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K/action/author_attestation","sign_citation":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K/action/citation_signature","submit_replication":"https://pith.science/pith/VXDOLB53C5Q4KHDEUJJK7CC55K/action/replication_record"}},"created_at":"2026-07-05T10:14:47.657526+00:00","updated_at":"2026-07-05T10:14:47.657526+00:00"}