{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:A5VPEZSIS3DY4X567LBCWAYS77","short_pith_number":"pith:A5VPEZSI","schema_version":"1.0","canonical_sha256":"076af2664896c78e5fbefac22b0312fffe447758df4e6e3f8fc407ff32ec18fd","source":{"kind":"arxiv","id":"2308.04624","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking LLM powered Chatbots: Methods and Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Arjun Avadhanam, Debarag Banerjee, Pooja Singh, Saksham Srivastava","submitted_at":"2023-08-08T23:30:20Z","abstract_excerpt":"Autonomous conversational agents, i.e. chatbots, are becoming an increasingly common mechanism for enterprises to provide support to customers and partners. In order to rate chatbots, especially ones powered by Generative AI tools like Large Language Models (LLMs) we need to be able to accurately assess their performance. This is where chatbot benchmarking becomes important. In this paper, we propose the use of a novel benchmark that we call the E2E (End to End) benchmark, and show how the E2E benchmark can be used to evaluate accuracy and usefulness of the answers provided by chatbots, especi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.04624","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-08T23:30:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5cf4ced1670542d1a1c6beb9f44ca8063d66f9082d13acc1a9d06ebca7cbd1bd","abstract_canon_sha256":"a56e2653462936670ce01f4230830d485989d90c274eff3495362a8f56346c8d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:39:40.472844Z","signature_b64":"nQn/7ZF0tsiexVqQcs51n0YQN12Wg5F9W3mhxVbOlY7BMyWR+8DGbvqy1qrDKfeMvVN2wWPlaNoPJiE9U2JSAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"076af2664896c78e5fbefac22b0312fffe447758df4e6e3f8fc407ff32ec18fd","last_reissued_at":"2026-07-05T06:39:40.472374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:39:40.472374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking LLM powered Chatbots: Methods and Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Arjun Avadhanam, Debarag Banerjee, Pooja Singh, Saksham Srivastava","submitted_at":"2023-08-08T23:30:20Z","abstract_excerpt":"Autonomous conversational agents, i.e. chatbots, are becoming an increasingly common mechanism for enterprises to provide support to customers and partners. In order to rate chatbots, especially ones powered by Generative AI tools like Large Language Models (LLMs) we need to be able to accurately assess their performance. This is where chatbot benchmarking becomes important. In this paper, we propose the use of a novel benchmark that we call the E2E (End to End) benchmark, and show how the E2E benchmark can be used to evaluate accuracy and usefulness of the answers provided by chatbots, especi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.04624","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.04624/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.04624","created_at":"2026-07-05T06:39:40.472432+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.04624v1","created_at":"2026-07-05T06:39:40.472432+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.04624","created_at":"2026-07-05T06:39:40.472432+00:00"},{"alias_kind":"pith_short_12","alias_value":"A5VPEZSIS3DY","created_at":"2026-07-05T06:39:40.472432+00:00"},{"alias_kind":"pith_short_16","alias_value":"A5VPEZSIS3DY4X56","created_at":"2026-07-05T06:39:40.472432+00:00"},{"alias_kind":"pith_short_8","alias_value":"A5VPEZSI","created_at":"2026-07-05T06:39:40.472432+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24834","citing_title":"Accuracy and Satisfaction in Multi-Turn LLM Dialogues for NFR Assessment","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01386","citing_title":"GuidaPA: Privacy-Preserving Chatbot for Public Administration via Federated Learning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17170","citing_title":"TriAxialKV: Toward Extreme Low-Precision KV-Cache Quantization for Agentic Inference Tasks","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2308.11432","citing_title":"A Survey on Large Language Model based Autonomous Agents","ref_index":174,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77","json":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77.json","graph_json":"https://pith.science/api/pith-number/A5VPEZSIS3DY4X567LBCWAYS77/graph.json","events_json":"https://pith.science/api/pith-number/A5VPEZSIS3DY4X567LBCWAYS77/events.json","paper":"https://pith.science/paper/A5VPEZSI"},"agent_actions":{"view_html":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77","download_json":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77.json","view_paper":"https://pith.science/paper/A5VPEZSI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.04624&json=true","fetch_graph":"https://pith.science/api/pith-number/A5VPEZSIS3DY4X567LBCWAYS77/graph.json","fetch_events":"https://pith.science/api/pith-number/A5VPEZSIS3DY4X567LBCWAYS77/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77/action/storage_attestation","attest_author":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77/action/author_attestation","sign_citation":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77/action/citation_signature","submit_replication":"https://pith.science/pith/A5VPEZSIS3DY4X567LBCWAYS77/action/replication_record"}},"created_at":"2026-07-05T06:39:40.472432+00:00","updated_at":"2026-07-05T06:39:40.472432+00:00"}