{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:UBWZXPN4J3K6EU5R7QKZN44A6C","short_pith_number":"pith:UBWZXPN4","schema_version":"1.0","canonical_sha256":"a06d9bbdbc4ed5e253b1fc1596f380f0b35851f765c4739206036d2aff04adcf","source":{"kind":"arxiv","id":"1909.10122","version":1},"attestation_state":"computed","paper":{"title":"Towards Best Experiment Design for Evaluating Dialogue System Output","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Samira Shaikh, Sashank Santhanam","submitted_at":"2019-09-23T01:45:55Z","abstract_excerpt":"To overcome the limitations of automated metrics (e.g. BLEU, METEOR) for evaluating dialogue systems, researchers typically use human judgments to provide convergent evidence. While it has been demonstrated that human judgments can suffer from the inconsistency of ratings, extant research has also found that the design of the evaluation task affects the consistency and quality of human judgments. We conduct a between-subjects study to understand the impact of four experiment conditions on human ratings of dialogue system output. In addition to discrete and continuous scale ratings, we also exp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.10122","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2019-09-23T01:45:55Z","cross_cats_sorted":[],"title_canon_sha256":"787f33a867512dc1d90d6bec6fd4c15f86eaeeedcb3081f6ab392b727d92e5cc","abstract_canon_sha256":"6af50a54ab460dcb348bd3acbb2b79c31e6311fc8692485eb5197952083279a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:06:17.840252Z","signature_b64":"TZrDNydWxqpOl4gSMJm4eTkh+URbHJ2/oLhddZ6TZkzh8O2GrAjNe1+WuDFb5MKxGXJp6dSh2qCWpYy8CgBjDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a06d9bbdbc4ed5e253b1fc1596f380f0b35851f765c4739206036d2aff04adcf","last_reissued_at":"2026-07-05T00:06:17.839793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:06:17.839793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Best Experiment Design for Evaluating Dialogue System Output","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Samira Shaikh, Sashank Santhanam","submitted_at":"2019-09-23T01:45:55Z","abstract_excerpt":"To overcome the limitations of automated metrics (e.g. BLEU, METEOR) for evaluating dialogue systems, researchers typically use human judgments to provide convergent evidence. While it has been demonstrated that human judgments can suffer from the inconsistency of ratings, extant research has also found that the design of the evaluation task affects the consistency and quality of human judgments. We conduct a between-subjects study to understand the impact of four experiment conditions on human ratings of dialogue system output. In addition to discrete and continuous scale ratings, we also exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.10122","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.10122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.10122","created_at":"2026-07-05T00:06:17.839856+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.10122v1","created_at":"2026-07-05T00:06:17.839856+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.10122","created_at":"2026-07-05T00:06:17.839856+00:00"},{"alias_kind":"pith_short_12","alias_value":"UBWZXPN4J3K6","created_at":"2026-07-05T00:06:17.839856+00:00"},{"alias_kind":"pith_short_16","alias_value":"UBWZXPN4J3K6EU5R","created_at":"2026-07-05T00:06:17.839856+00:00"},{"alias_kind":"pith_short_8","alias_value":"UBWZXPN4","created_at":"2026-07-05T00:06:17.839856+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C","json":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C.json","graph_json":"https://pith.science/api/pith-number/UBWZXPN4J3K6EU5R7QKZN44A6C/graph.json","events_json":"https://pith.science/api/pith-number/UBWZXPN4J3K6EU5R7QKZN44A6C/events.json","paper":"https://pith.science/paper/UBWZXPN4"},"agent_actions":{"view_html":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C","download_json":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C.json","view_paper":"https://pith.science/paper/UBWZXPN4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.10122&json=true","fetch_graph":"https://pith.science/api/pith-number/UBWZXPN4J3K6EU5R7QKZN44A6C/graph.json","fetch_events":"https://pith.science/api/pith-number/UBWZXPN4J3K6EU5R7QKZN44A6C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C/action/storage_attestation","attest_author":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C/action/author_attestation","sign_citation":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C/action/citation_signature","submit_replication":"https://pith.science/pith/UBWZXPN4J3K6EU5R7QKZN44A6C/action/replication_record"}},"created_at":"2026-07-05T00:06:17.839856+00:00","updated_at":"2026-07-05T00:06:17.839856+00:00"}