{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B3JCC5KLLHOUCJNZA4WEPWVXFD","short_pith_number":"pith:B3JCC5KL","schema_version":"1.0","canonical_sha256":"0ed221754b59dd4125b9072c47dab728dc3b4c77d68a785bac17d031d66c721e","source":{"kind":"arxiv","id":"2404.01015","version":2},"attestation_state":"computed","paper":{"title":"PairEval: Open-domain Dialogue Evaluation with Pairwise Comparison","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"ChaeHun Park, Dohyun Lee, Jaegul Choo, Minseok Choi","submitted_at":"2024-04-01T09:35:06Z","abstract_excerpt":"Building a reliable and automated evaluation metric is a necessary but challenging problem for open-domain dialogue systems. Recent studies proposed evaluation metrics that assess generated responses by considering their relevance to previous dialogue histories. Although effective, these metrics evaluate individual responses directly rather than considering their relative quality compared to other responses. To handle this, we propose PairEval, a novel dialogue evaluation metric for assessing responses by comparing their quality against responses in different conversations. PairEval is built o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.01015","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-01T09:35:06Z","cross_cats_sorted":[],"title_canon_sha256":"edfe0da704afcad8121a7d39190ae07f097f519c591446b4575db3e30c714644","abstract_canon_sha256":"2ae1e6417173ca945055b38a6b02fbb20613522bf26446773240e5faaf292ffe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:27.112153Z","signature_b64":"zy4PjJgZBAPdGo8anY3rkOhq8FWhH2BveVkRz+wu4n6DF4n6n5gy0KfH3bys+161W6vLg66avbDUhudTRut3DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ed221754b59dd4125b9072c47dab728dc3b4c77d68a785bac17d031d66c721e","last_reissued_at":"2026-07-05T08:45:27.111663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:27.111663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PairEval: Open-domain Dialogue Evaluation with Pairwise Comparison","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"ChaeHun Park, Dohyun Lee, Jaegul Choo, Minseok Choi","submitted_at":"2024-04-01T09:35:06Z","abstract_excerpt":"Building a reliable and automated evaluation metric is a necessary but challenging problem for open-domain dialogue systems. Recent studies proposed evaluation metrics that assess generated responses by considering their relevance to previous dialogue histories. Although effective, these metrics evaluate individual responses directly rather than considering their relative quality compared to other responses. To handle this, we propose PairEval, a novel dialogue evaluation metric for assessing responses by comparing their quality against responses in different conversations. PairEval is built o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01015","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.01015","created_at":"2026-07-05T08:45:27.111725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.01015v2","created_at":"2026-07-05T08:45:27.111725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01015","created_at":"2026-07-05T08:45:27.111725+00:00"},{"alias_kind":"pith_short_12","alias_value":"B3JCC5KLLHOU","created_at":"2026-07-05T08:45:27.111725+00:00"},{"alias_kind":"pith_short_16","alias_value":"B3JCC5KLLHOUCJNZ","created_at":"2026-07-05T08:45:27.111725+00:00"},{"alias_kind":"pith_short_8","alias_value":"B3JCC5KL","created_at":"2026-07-05T08:45:27.111725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.15240","citing_title":"Generalised Probabilistic Modelling and Improved Uncertainty Estimation in Comparative LLM-as-a-judge","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD","json":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD.json","graph_json":"https://pith.science/api/pith-number/B3JCC5KLLHOUCJNZA4WEPWVXFD/graph.json","events_json":"https://pith.science/api/pith-number/B3JCC5KLLHOUCJNZA4WEPWVXFD/events.json","paper":"https://pith.science/paper/B3JCC5KL"},"agent_actions":{"view_html":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD","download_json":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD.json","view_paper":"https://pith.science/paper/B3JCC5KL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.01015&json=true","fetch_graph":"https://pith.science/api/pith-number/B3JCC5KLLHOUCJNZA4WEPWVXFD/graph.json","fetch_events":"https://pith.science/api/pith-number/B3JCC5KLLHOUCJNZA4WEPWVXFD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD/action/storage_attestation","attest_author":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD/action/author_attestation","sign_citation":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD/action/citation_signature","submit_replication":"https://pith.science/pith/B3JCC5KLLHOUCJNZA4WEPWVXFD/action/replication_record"}},"created_at":"2026-07-05T08:45:27.111725+00:00","updated_at":"2026-07-05T08:45:27.111725+00:00"}