{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:JJCAMTDR7UOIYXMI3672R5E2VP","short_pith_number":"pith:JJCAMTDR","schema_version":"1.0","canonical_sha256":"4a44064c71fd1c8c5d88dfbfa8f49aabde125c1829921fcb4656c4a2c2aae5ce","source":{"kind":"arxiv","id":"2108.01369","version":1},"attestation_state":"computed","paper":{"title":"How to Evaluate Your Dialogue Models: A Review of Approaches","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Long Qin, Quanjun Yin, Wansen Wu, Xinmeng Li","submitted_at":"2021-08-03T08:52:33Z","abstract_excerpt":"Evaluating the quality of a dialogue system is an understudied problem. The recent evolution of evaluation method motivated this survey, in which an explicit and comprehensive analysis of the existing methods is sought. We are first to divide the evaluation methods into three classes, i.e., automatic evaluation, human-involved evaluation and user simulator based evaluation. Then, each class is covered with main features and the related evaluation metrics. The existence of benchmarks, suitable for the evaluation of dialogue techniques are also discussed in detail. Finally, some open issues are "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.01369","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-08-03T08:52:33Z","cross_cats_sorted":[],"title_canon_sha256":"5714d7262f0bf2cf0ce5ce17ecc8c900fbbc10fd8b6123c626987bd64bbad207","abstract_canon_sha256":"fe4c26b18500a0e34cf84250873eb50dd16c9291ebe7b5a573c5079029e453a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:03:02.000133Z","signature_b64":"Pq27mBZb0RrDGQMq9A8IVg+36R37dGjjkFhIMMfYdMR03gQ1rYdbEIlCkJMNVDpXVoaV2H6deZ4TAJsNITbHCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a44064c71fd1c8c5d88dfbfa8f49aabde125c1829921fcb4656c4a2c2aae5ce","last_reissued_at":"2026-07-05T03:03:01.999657Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:03:01.999657Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Evaluate Your Dialogue Models: A Review of Approaches","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Long Qin, Quanjun Yin, Wansen Wu, Xinmeng Li","submitted_at":"2021-08-03T08:52:33Z","abstract_excerpt":"Evaluating the quality of a dialogue system is an understudied problem. The recent evolution of evaluation method motivated this survey, in which an explicit and comprehensive analysis of the existing methods is sought. We are first to divide the evaluation methods into three classes, i.e., automatic evaluation, human-involved evaluation and user simulator based evaluation. Then, each class is covered with main features and the related evaluation metrics. The existence of benchmarks, suitable for the evaluation of dialogue techniques are also discussed in detail. Finally, some open issues are "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.01369","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.01369/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.01369","created_at":"2026-07-05T03:03:01.999722+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.01369v1","created_at":"2026-07-05T03:03:01.999722+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.01369","created_at":"2026-07-05T03:03:01.999722+00:00"},{"alias_kind":"pith_short_12","alias_value":"JJCAMTDR7UOI","created_at":"2026-07-05T03:03:01.999722+00:00"},{"alias_kind":"pith_short_16","alias_value":"JJCAMTDR7UOIYXMI","created_at":"2026-07-05T03:03:01.999722+00:00"},{"alias_kind":"pith_short_8","alias_value":"JJCAMTDR","created_at":"2026-07-05T03:03:01.999722+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.17348","citing_title":"Better Slow than Sorry: Introducing Positive Friction for Reliable Dialogue Systems","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP","json":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP.json","graph_json":"https://pith.science/api/pith-number/JJCAMTDR7UOIYXMI3672R5E2VP/graph.json","events_json":"https://pith.science/api/pith-number/JJCAMTDR7UOIYXMI3672R5E2VP/events.json","paper":"https://pith.science/paper/JJCAMTDR"},"agent_actions":{"view_html":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP","download_json":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP.json","view_paper":"https://pith.science/paper/JJCAMTDR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.01369&json=true","fetch_graph":"https://pith.science/api/pith-number/JJCAMTDR7UOIYXMI3672R5E2VP/graph.json","fetch_events":"https://pith.science/api/pith-number/JJCAMTDR7UOIYXMI3672R5E2VP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP/action/storage_attestation","attest_author":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP/action/author_attestation","sign_citation":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP/action/citation_signature","submit_replication":"https://pith.science/pith/JJCAMTDR7UOIYXMI3672R5E2VP/action/replication_record"}},"created_at":"2026-07-05T03:03:01.999722+00:00","updated_at":"2026-07-05T03:03:01.999722+00:00"}