{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:KIKJF4SFLK4XEKV3WZGOCI6ZPH","short_pith_number":"pith:KIKJF4SF","schema_version":"1.0","canonical_sha256":"521492f2455ab9722abbb64ce123d979f36c145a4c72bc1ab195203ee6e844a1","source":{"kind":"arxiv","id":"2105.14488","version":2},"attestation_state":"computed","paper":{"title":"REAM$\\sharp$: An Enhancement Approach to Reference-based Evaluation Metrics for Open-domain Dialog Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jun Gao, Ruifeng Xu, Shuming Shi, Wei Bi","submitted_at":"2021-05-30T10:04:13Z","abstract_excerpt":"The lack of reliable automatic evaluation metrics is a major impediment to the development of open-domain dialogue systems. Various reference-based metrics have been proposed to calculate a score between a predicted response and a small set of references. However, these metrics show unsatisfactory correlations with human judgments. For a reference-based metric, its reliability mainly depends on two factors: its ability to measure the similarity between the predicted response and the reference response, as well as the reliability of the given reference set. Yet, there are few discussions on the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.14488","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-05-30T10:04:13Z","cross_cats_sorted":[],"title_canon_sha256":"8e60fffe183e3df092c6ef4971fdc3e5e4ca34ebcdfbbbda7b041f8b6465fa74","abstract_canon_sha256":"331736089e084adc826df8b0faf0b713661e73c6c323b9b8f782f943d3ad240e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:31.264452Z","signature_b64":"Q2iLN5lgSymSARoOVlQX/x3/0P7a9m8gp/9BO1ND+AnpjeaaMjDtDhJoZjjKbhX+atLKQP/MU2sYx8xo4IESCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"521492f2455ab9722abbb64ce123d979f36c145a4c72bc1ab195203ee6e844a1","last_reissued_at":"2026-07-05T04:05:31.263854Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:31.263854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"REAM$\\sharp$: An Enhancement Approach to Reference-based Evaluation Metrics for Open-domain Dialog Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jun Gao, Ruifeng Xu, Shuming Shi, Wei Bi","submitted_at":"2021-05-30T10:04:13Z","abstract_excerpt":"The lack of reliable automatic evaluation metrics is a major impediment to the development of open-domain dialogue systems. Various reference-based metrics have been proposed to calculate a score between a predicted response and a small set of references. However, these metrics show unsatisfactory correlations with human judgments. For a reference-based metric, its reliability mainly depends on two factors: its ability to measure the similarity between the predicted response and the reference response, as well as the reliability of the given reference set. Yet, there are few discussions on the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.14488","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.14488/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.14488","created_at":"2026-07-05T04:05:31.263911+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.14488v2","created_at":"2026-07-05T04:05:31.263911+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.14488","created_at":"2026-07-05T04:05:31.263911+00:00"},{"alias_kind":"pith_short_12","alias_value":"KIKJF4SFLK4X","created_at":"2026-07-05T04:05:31.263911+00:00"},{"alias_kind":"pith_short_16","alias_value":"KIKJF4SFLK4XEKV3","created_at":"2026-07-05T04:05:31.263911+00:00"},{"alias_kind":"pith_short_8","alias_value":"KIKJF4SF","created_at":"2026-07-05T04:05:31.263911+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH","json":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH.json","graph_json":"https://pith.science/api/pith-number/KIKJF4SFLK4XEKV3WZGOCI6ZPH/graph.json","events_json":"https://pith.science/api/pith-number/KIKJF4SFLK4XEKV3WZGOCI6ZPH/events.json","paper":"https://pith.science/paper/KIKJF4SF"},"agent_actions":{"view_html":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH","download_json":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH.json","view_paper":"https://pith.science/paper/KIKJF4SF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.14488&json=true","fetch_graph":"https://pith.science/api/pith-number/KIKJF4SFLK4XEKV3WZGOCI6ZPH/graph.json","fetch_events":"https://pith.science/api/pith-number/KIKJF4SFLK4XEKV3WZGOCI6ZPH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH/action/storage_attestation","attest_author":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH/action/author_attestation","sign_citation":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH/action/citation_signature","submit_replication":"https://pith.science/pith/KIKJF4SFLK4XEKV3WZGOCI6ZPH/action/replication_record"}},"created_at":"2026-07-05T04:05:31.263911+00:00","updated_at":"2026-07-05T04:05:31.263911+00:00"}